From f75c30f78da9d844f5207b6d272d854aed65b23f Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Mon, 14 Apr 2025 16:45:13 +0200 Subject: [PATCH 01/56] fix(test): Link to Boost::boost for all installed header-only libs --- test/CMakeLists.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/test/CMakeLists.txt b/test/CMakeLists.txt index d5378540..9618627f 100644 --- a/test/CMakeLists.txt +++ b/test/CMakeLists.txt @@ -27,7 +27,7 @@ find_package(Boost REQUIRED) add_executable(GPRat_test_output_correctness src/output_correctness.cpp) target_link_libraries(GPRat_test_output_correctness - PRIVATE GPRat::core Catch2::Catch2WithMain) + PRIVATE GPRat::core Catch2::Catch2WithMain Boost::boost) target_compile_features(GPRat_test_output_correctness PRIVATE cxx_std_17) add_test( From f88e807fa1bd15152f5cf5a34a0cee76f6b0c5a5 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Mon, 14 Apr 2025 16:47:07 +0200 Subject: [PATCH 02/56] chore(core): Make code more portable - Don't rely on nonstandard extensions (such as M_PI) - Don't rely on transitive includes for some standard types - Switch to C++20 so we can use std::numbers --- core/CMakeLists.txt | 7 ++----- core/src/cpu/gp_algorithms.cpp | 1 + core/src/cpu/gp_optimizer.cpp | 3 ++- core/src/gp_hyperparameters.cpp | 1 + 4 files changed, 6 insertions(+), 6 deletions(-) diff --git a/core/CMakeLists.txt b/core/CMakeLists.txt index da4c96d0..2ee98d07 100644 --- a/core/CMakeLists.txt +++ b/core/CMakeLists.txt @@ -66,16 +66,13 @@ if(GPRAT_ENABLE_MKL) # Link Intel oneMKL target_link_libraries(gprat_core PUBLIC MKL::mkl_intel_lp64 MKL::mkl_core MKL::MKL MKL::mkl_sequential) + target_compile_definitions(gprat_core PUBLIC GPRAT_ENABLE_MKL) else() # Link OpenBLAS target_link_libraries(gprat_core PUBLIC ${OpenBLAS_LIB}) endif() -if(GPRAT_ENABLE_MKL) - target_compile_definitions(gprat_core PUBLIC GPRAT_ENABLE_MKL) -endif() - -target_compile_features(gprat_core PUBLIC cxx_std_17) +target_compile_features(gprat_core PUBLIC cxx_std_20) set_property(TARGET gprat_core PROPERTY POSITION_INDEPENDENT_CODE ON) diff --git a/core/src/cpu/gp_algorithms.cpp b/core/src/cpu/gp_algorithms.cpp index 95eb2e2f..92193b6d 100644 --- a/core/src/cpu/gp_algorithms.cpp +++ b/core/src/cpu/gp_algorithms.cpp @@ -1,6 +1,7 @@ #include "cpu/gp_algorithms.hpp" #include +#include namespace cpu { diff --git a/core/src/cpu/gp_optimizer.cpp b/core/src/cpu/gp_optimizer.cpp index f9c5d500..d33b1889 100644 --- a/core/src/cpu/gp_optimizer.cpp +++ b/core/src/cpu/gp_optimizer.cpp @@ -1,6 +1,7 @@ #include "cpu/gp_optimizer.hpp" #include "cpu/adapter_cblas_fp64.hpp" +#include #include namespace cpu @@ -212,7 +213,7 @@ double add_losses(const std::vector &losses, std::size_t N, std::size_t l += losses[i]; } - l += Nn * log(2.0 * M_PI); + l += Nn * log(2.0 * std::numbers::pi); return 0.5 * l / Nn; // why /Nn? } diff --git a/core/src/gp_hyperparameters.cpp b/core/src/gp_hyperparameters.cpp index c7c0d9c0..f0a8caab 100644 --- a/core/src/gp_hyperparameters.cpp +++ b/core/src/gp_hyperparameters.cpp @@ -1,6 +1,7 @@ #include "gp_hyperparameters.hpp" #include +#include namespace gprat_hyper { From 6a68745c81c4416a9ed249ab15e0fe814dc0e8f1 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Tue, 15 Apr 2025 23:06:42 +0200 Subject: [PATCH 03/56] feat: Add vcpkg as an alternative to spack (mainly for Windows) --- CMakePresets.json | 19 +++++++++++++++++-- external_ports/README.md | 3 +++ vcpkg-configuration.json | 6 ++++++ vcpkg.json | 21 +++++++++++++++++++++ 4 files changed, 47 insertions(+), 2 deletions(-) create mode 100644 external_ports/README.md create mode 100644 vcpkg-configuration.json create mode 100644 vcpkg.json diff --git a/CMakePresets.json b/CMakePresets.json index e18ab19b..95204452 100644 --- a/CMakePresets.json +++ b/CMakePresets.json @@ -21,6 +21,21 @@ "deprecated": true } }, + { + "name": "vcpkg", + "hidden": true, + "cacheVariables": { + "CMAKE_TOOLCHAIN_FILE": "$env{VCPKG_ROOT}/scripts/buildsystems/vcpkg.cmake", + "X_VCPKG_APPLOCAL_DEPS_INSTALL": "ON" + } + }, + { + "name": "vcpkg-win64-static", + "hidden": true, + "cacheVariables": { + "VCPKG_TARGET_TRIPLET": "x64-windows-static-md-release" + } + }, { "name": "cppcheck", "hidden": true, @@ -67,7 +82,7 @@ "description": "Note that all the flags after /W4 are required for MSVC to conform to the language standard", "hidden": true, "cacheVariables": { - "CMAKE_CXX_FLAGS": "/sdl /guard:cf /utf-8 /diagnostics:caret /w14165 /w44242 /w44254 /w44263 /w34265 /w34287 /w44296 /w44365 /w44388 /w44464 /w14545 /w14546 /w14547 /w14549 /w14555 /w34619 /w34640 /w24826 /w14905 /w14906 /w14928 /w45038 /W4 /permissive- /volatile:iso /Zc:inline /Zc:preprocessor /Zc:enumTypes /Zc:lambda /Zc:__cplusplus /Zc:externConstexpr /Zc:throwingNew /EHsc", + "CMAKE_CXX_FLAGS": "/sdl /guard:cf /utf-8 /diagnostics:caret /w14165 /w44242 /w44254 /w44263 /w34265 /w34287 /w44296 /w44365 /w44388 /w44464 /w14545 /w14546 /w14547 /w14549 /w14555 /w34619 /w34640 /w24826 /w14905 /w14906 /w14928 /w45038 /W4 /permissive- /volatile:iso /Zc:inline /Zc:preprocessor /Zc:enumTypes /Zc:lambda /Zc:__cplusplus /Zc:externConstexpr /Zc:throwingNew /EHsc /D_CRT_SECURE_NO_WARNINGS", "CMAKE_EXE_LINKER_FLAGS": "/machine:x64 /guard:cf", "CMAKE_SHARED_LINKER_FLAGS": "/machine:x64 /guard:cf" } @@ -146,7 +161,7 @@ }, { "name": "ci-windows", - "inherits": ["ci-build", "ci-win64", "ci-multi-config"] + "inherits": ["ci-build", "ci-win64", "ci-multi-config", "vcpkg", "vcpkg-win64-static"] }, { "name": "ci-ubuntu-24.04", diff --git a/external_ports/README.md b/external_ports/README.md new file mode 100644 index 00000000..993ec19c --- /dev/null +++ b/external_ports/README.md @@ -0,0 +1,3 @@ +# What is this? + +This contains custom vcpkg ports and forks of official ones. diff --git a/vcpkg-configuration.json b/vcpkg-configuration.json new file mode 100644 index 00000000..3afcbd70 --- /dev/null +++ b/vcpkg-configuration.json @@ -0,0 +1,6 @@ +{ + "$schema": "https://raw.githubusercontent.com/microsoft/vcpkg-tool/main/docs/vcpkg-configuration.schema.json", + "overlay-ports": [ + "./external_ports" + ] +} diff --git a/vcpkg.json b/vcpkg.json new file mode 100644 index 00000000..438621a2 --- /dev/null +++ b/vcpkg.json @@ -0,0 +1,21 @@ +{ + "$schema": "https://raw.githubusercontent.com/microsoft/vcpkg-tool/main/docs/vcpkg.schema.json", + "name": "gprat", + "version-semver": "0.1.0", + "dependencies": [ + { + "name": "boost-json" + }, + { + "name": "intel-mkl" + }, + { + "name": "fmt" + }, + { + "name": "hpx" + } + ], + "default-features": [], + "builtin-baseline": "e08b7bd89ae162f8579df2f8d39a1ae94107c8fd" +} From 50809bb873d0d182c72b19c8ddff48729151dff9 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Tue, 15 Apr 2025 22:40:55 +0200 Subject: [PATCH 04/56] chore(vcpkg): Import vanilla intel-mkl port from 6b575523ce838fc13517d1a8021ce4883efc29c1 --- external_ports/intel-mkl/copy-from-dmg.cmake | 53 ++++ external_ports/intel-mkl/portfile.cmake | 260 +++++++++++++++++++ external_ports/intel-mkl/usage | 4 + external_ports/intel-mkl/vcpkg.json | 16 ++ 4 files changed, 333 insertions(+) create mode 100644 external_ports/intel-mkl/copy-from-dmg.cmake create mode 100644 external_ports/intel-mkl/portfile.cmake create mode 100644 external_ports/intel-mkl/usage create mode 100644 external_ports/intel-mkl/vcpkg.json diff --git a/external_ports/intel-mkl/copy-from-dmg.cmake b/external_ports/intel-mkl/copy-from-dmg.cmake new file mode 100644 index 00000000..a5aa67cd --- /dev/null +++ b/external_ports/intel-mkl/copy-from-dmg.cmake @@ -0,0 +1,53 @@ +find_program(HDIUTIL NAMES hdiutil REQUIRED) +set(dmg_path "NOTFOUND" CACHE FILEPATH "Where to find the DMG") +set(output_dir "output_dir" CACHE FILEPATH "Where to put the packages") + +if(NOT EXISTS "${dmg_path}") + message(FATAL_ERROR "'dmg_path' (${dmg_path}) does not exist.") +endif() +if(NOT IS_DIRECTORY "${output_dir}") + message(FATAL_ERROR "'output_dir' (${output_dir}) is not a directory.") +endif() + +execute_process( + COMMAND mktemp -d + RESULT_VARIABLE mktemp_result + OUTPUT_VARIABLE mount_point + OUTPUT_STRIP_TRAILING_WHITESPACE +) +if(NOT mktemp_result STREQUAL "0") + message(FATAL_ERROR "mktemp -d failed: ${mktemp_result}") +elseif(NOT IS_DIRECTORY "${mount_point}") + message(FATAL_ERROR "'mount_point' (${mount_point}) is not a directory.") +endif() + +execute_process( + COMMAND "${HDIUTIL}" attach "${dmg_path}" -mountpoint "${mount_point}" -readonly + RESULT_VARIABLE mount_result +) +if(mount_result STREQUAL "0") + set(dmg_packages_dir "${mount_point}/bootstrapper.app/Contents/Resources/packages") + file(GLOB packages + "${dmg_packages_dir}/intel.oneapi.mac.mkl.devel,*" + "${dmg_packages_dir}/intel.oneapi.mac.mkl.runtime,*" + "${dmg_packages_dir}/intel.oneapi.mac.mkl.product,*" + "${dmg_packages_dir}/intel.oneapi.mac.openmp,*" + ) + # Using execute_process to avoid direct errors + execute_process( + COMMAND cp -R ${packages} "${output_dir}/" + RESULT_VARIABLE copy_result + ) +endif() +execute_process( + COMMAND "${HDIUTIL}" detach "${mount_point}" + RESULT_VARIABLE unmount_result +) + +if(NOT mount_result STREQUAL "0") + message(FATAL_ERROR "Mounting ${dmg_path} failed: ${mount_result}") +elseif(NOT copy_result STREQUAL "0") + message(FATAL_ERROR "Coyping packages failed: ${copy_result}") +elseif(NOT unmount_result STREQUAL "0") + message(FATAL_ERROR "Unounting ${dmg_path} failed: ${unmount_result}") +endif() diff --git a/external_ports/intel-mkl/portfile.cmake b/external_ports/intel-mkl/portfile.cmake new file mode 100644 index 00000000..908a5281 --- /dev/null +++ b/external_ports/intel-mkl/portfile.cmake @@ -0,0 +1,260 @@ +# This package installs Intel MKL on Linux, macOS and Windows for x64. +# Configuration: +# - ilp64 +# - dynamic CRT: intel_thread, static CRT: sequential + +set(VCPKG_POLICY_EMPTY_PACKAGE enabled) + +# https://registrationcenter-download.intel.com/akdlm/IRC_NAS/19150/w_onemkl_p_2023.0.0.25930_offline.exe # windows +# https://registrationcenter-download.intel.com/akdlm/IRC_NAS/19116/m_onemkl_p_2023.0.0.25376_offline.dmg # macos +# https://registrationcenter-download.intel.com/akdlm/irc_nas/19138/l_onemkl_p_2023.0.0.25398_offline.sh # linux +set(sha "") +if(NOT VCPKG_TARGET_ARCHITECTURE STREQUAL "x64") + # nop +elseif(VCPKG_TARGET_IS_WINDOWS) + set(filename w_onemkl_p_2023.0.0.25930_offline.exe) + set(magic_number 19150) + set(sha a3eb6b75241a2eccb73ed73035ff111172c55d3fa51f545c7542277a155df84ff72fc826621711153e683f84058e64cb549c030968f9f964531db76ca8a3ed46) + set(package_infix "win") +elseif(VCPKG_TARGET_IS_OSX) + set(filename m_onemkl_p_2023.0.0.25376_offline.dmg) + set(magic_number 19116) + set(sha 7b9b8c004054603e6830fb9b9c049d5a4cfc0990c224cb182ac5262ab9f1863775a67491413040e3349c590e2cca58edcfc704db9f3b9f9faa8b5b09022cd2af) + set(package_infix "mac") + set(package_libdir "lib") + set(compiler_libdir "mac/compiler/lib") +elseif(VCPKG_TARGET_IS_LINUX) + set(filename l_onemkl_p_2023.0.0.25398_offline.sh) + set(magic_number 19138) + set(sha b5f2f464675f0fd969dde2faf2e622b834eb1cc406c4a867148116f6c24ba5c709d98b678840f4a89a1778e12cde0ff70ce2ef59faeef3d3f3aa1d0329c71af1) + set(package_infix "lin") + set(package_libdir "lib/intel64") + set(compiler_libdir "linux/compiler/lib/intel64_lin") +endif() + +if(NOT sha) + message(WARNING "${PORT} is empty for ${TARGET_TRIPLET}.") + return() +endif() + +vcpkg_download_distfile(installer_path + URLS "https://registrationcenter-download.intel.com/akdlm/IRC_NAS/${magic_number}/${filename}" + FILENAME "${filename}" + SHA512 "${sha}" +) + +# Note: intel_thread and lp64 are the defaults. +set(interface "ilp64") # or ilp64; ilp == 64 bit int api +#https://www.intel.com/content/www/us/en/develop/documentation/onemkl-linux-developer-guide/top/linking-your-application-with-onemkl/linking-in-detail/linking-with-interface-libraries/using-the-ilp64-interface-vs-lp64-interface.html +if(VCPKG_CRT_LINKAGE STREQUAL "dynamic") + set(threading "intel_thread") #sequential or intel_thread or tbb_thread or pgi_thread +else() + set(threading "sequential") +endif() +if(threading STREQUAL "intel_thread") + set(short_thread "iomp") +else() + string(SUBSTRING "${threading}" "0" "3" short_thread) +endif() +set(main_pc_file "mkl-${VCPKG_LIBRARY_LINKAGE}-${interface}-${short_thread}.pc") + +# First extraction level: packages (from offline installer) +set(extract_0_dir "${CURRENT_BUILDTREES_DIR}/${TARGET_TRIPLET}-extract") +file(REMOVE_RECURSE "${extract_0_dir}") +file(MAKE_DIRECTORY "${extract_0_dir}") + +# Second extraction level: actual files (from packages) +set(extract_1_dir "${CURRENT_PACKAGES_DIR}/intel-extract") +file(REMOVE_RECURSE "${extract_1_dir}") +file(MAKE_DIRECTORY "${extract_1_dir}") + +file(MAKE_DIRECTORY "${CURRENT_PACKAGES_DIR}/lib/pkgconfig") + +if(VCPKG_TARGET_IS_WINDOWS) + vcpkg_find_acquire_program(7Z) + message(STATUS "Extracting offline installer") + vcpkg_execute_required_process( + COMMAND "${7Z}" x "${installer_path}" "-o${extract_0_dir}" "-y" "-bso0" "-bsp0" + WORKING_DIRECTORY "${extract_0_dir}" + LOGNAME "extract-${TARGET_TRIPLET}-0" + ) + + set(packages + "intel.oneapi.win.mkl.devel,v=2023.0.0-25930/oneapi-mkl-devel-for-installer_p_2023.0.0.25930.msi" # has the required libs. + "intel.oneapi.win.mkl.runtime,v=2023.0.0-25930/oneapi-mkl-for-installer_p_2023.0.0.25930.msi" # has the required DLLs + #"intel.oneapi.win.compilers-common-runtime,v=2023.0.0-25922" # SVML + "intel.oneapi.win.openmp,v=2023.0.0-25922/oneapi-comp-openmp-for-installer_p_2023.0.0.25922.msi" # OpenMP + #"intel.oneapi.win.tbb.runtime,v=2021.8.0-25874" #TBB + ) + + foreach(pack IN LISTS packages) + set(package_path "${extract_0_dir}/packages/${pack}") + cmake_path(GET pack STEM LAST_ONLY packstem) + cmake_path(NATIVE_PATH package_path package_path_native) + vcpkg_execute_required_process( + COMMAND "${LESSMSI}" x "${package_path_native}" + WORKING_DIRECTORY "${extract_1_dir}" + LOGNAME "extract-${TARGET_TRIPLET}-${packstem}" + ) + file(COPY "${extract_1_dir}/${packstem}/SourceDir/" DESTINATION "${extract_1_dir}") + file(REMOVE_RECURSE "${extract_1_dir}/${packstem}") + endforeach() + + set(mkl_dir "${extract_1_dir}/Intel/Compiler/12.0/mkl/2023.0.0") + file(COPY "${mkl_dir}/include/" DESTINATION "${CURRENT_PACKAGES_DIR}/include") + # see https://www.intel.com/content/www/us/en/developer/tools/oneapi/onemkl-link-line-advisor.html for linking + if(VCPKG_LIBRARY_LINKAGE STREQUAL "dynamic") + set(files "mkl_core_dll.lib" "mkl_${threading}_dll.lib" "mkl_intel_${interface}_dll.lib" "mkl_blas95_${interface}.lib" "mkl_lapack95_${interface}.lib") # "mkl_rt.lib" single dynamic lib with dynamic dispatch + file(COPY "${mkl_dir}/redist/intel64/" DESTINATION "${CURRENT_PACKAGES_DIR}/bin") # Could probably be reduced instead of copying all + if(NOT VCPKG_BUILD_TYPE) + file(COPY "${mkl_dir}/redist/intel64/" DESTINATION "${CURRENT_PACKAGES_DIR}/debug/bin") + endif() + else() + set(files "mkl_core.lib" "mkl_${threading}.lib" "mkl_intel_${interface}.lib" "mkl_blas95_${interface}.lib" "mkl_lapack95_${interface}.lib") + endif() + foreach(file IN LISTS files) + file(COPY "${mkl_dir}/lib/intel64/${file}" DESTINATION "${CURRENT_PACKAGES_DIR}/lib/intel64") # instead of manual-link keep normal structure + if(NOT VCPKG_BUILD_TYPE) + file(COPY "${mkl_dir}/lib/intel64/${file}" DESTINATION "${CURRENT_PACKAGES_DIR}/debug/lib/intel64") + endif() + endforeach() + file(COPY_FILE "${mkl_dir}/lib/pkgconfig/${main_pc_file}" "${CURRENT_PACKAGES_DIR}/lib/pkgconfig/${main_pc_file}") + + set(compiler_dir "${extract_1_dir}/Intel/Compiler/12.0/compiler/2023.0.0") + if(threading STREQUAL "intel_thread") + file(COPY "${compiler_dir}/windows/redist/intel64_win/compiler/" DESTINATION "${CURRENT_PACKAGES_DIR}/bin") + file(COPY "${compiler_dir}/windows/compiler/lib/intel64_win/" DESTINATION "${CURRENT_PACKAGES_DIR}/lib/intel64") + file(COPY_FILE "${compiler_dir}/lib/pkgconfig/openmp.pc" "${CURRENT_PACKAGES_DIR}/lib/pkgconfig/libiomp5.pc") + vcpkg_replace_string("${CURRENT_PACKAGES_DIR}/lib/pkgconfig/libiomp5.pc" "/windows/compiler/lib/intel64_win/" "/lib/intel64/") + vcpkg_replace_string("${CURRENT_PACKAGES_DIR}/lib/pkgconfig/libiomp5.pc" "-I \${includedir}" "-I\"\${includedir}\"") + vcpkg_replace_string("${CURRENT_PACKAGES_DIR}/lib/pkgconfig/${main_pc_file}" "openmp" "libiomp5") + if(NOT VCPKG_BUILD_TYPE) + file(COPY "${compiler_dir}/windows/redist/intel64_win/compiler/" DESTINATION "${CURRENT_PACKAGES_DIR}/debug/bin") + file(COPY "${compiler_dir}/windows/compiler/lib/intel64_win/" DESTINATION "${CURRENT_PACKAGES_DIR}/debug/lib/intel64") + endif() + endif() +else() + message(STATUS "Warning: This port is still a work on progress. + E.g. it is not correctly filtering the libraries in accordance with + VCPKG_LIBRARY_LINKAGE. It is using the default threading (Intel OpenMP) + which is known to segfault when used together with GNU OpenMP. +") + + message(STATUS "Extracting offline installer") + if(VCPKG_TARGET_IS_LINUX) + vcpkg_execute_required_process( + COMMAND "bash" "--verbose" "--noprofile" "${installer_path}" "--extract-only" "--extract-folder" "${extract_0_dir}" + WORKING_DIRECTORY "${extract_0_dir}" + LOGNAME "extract-${TARGET_TRIPLET}-0" + ) + file(RENAME "${extract_0_dir}/l_onemkl_p_2023.0.0.25398_offline/packages" "${extract_0_dir}/packages") + elseif(VCPKG_TARGET_IS_OSX) + find_program(HDIUTIL NAMES hdiutil REQUIRED) + file(MAKE_DIRECTORY "${extract_0_dir}/packages") + message(STATUS "... Don't interrupt.") + vcpkg_execute_required_process( + COMMAND "${CMAKE_COMMAND}" "-Ddmg_path=${installer_path}" + "-Doutput_dir=${extract_0_dir}/packages" + "-DHDIUTIL=${HDIUTIL}" + -P "${CMAKE_CURRENT_LIST_DIR}/copy-from-dmg.cmake" + WORKING_DIRECTORY "${extract_0_dir}" + LOGNAME "extract-${TARGET_TRIPLET}-0" + ) + message(STATUS "... Done.") + endif() + + file(GLOB package_path "${extract_0_dir}/packages/intel.oneapi.${package_infix}.mkl.runtime,v=2023.0.0-*") + cmake_path(GET package_path STEM LAST_ONLY packstem) + message(STATUS "Extracting ${packstem}") + vcpkg_execute_required_process( + COMMAND "${CMAKE_COMMAND}" "-E" "tar" "-xf" "${package_path}/cupPayload.cup" + "_installdir/mkl/2023.0.0/lib" + "_installdir/mkl/2023.0.0/licensing" + WORKING_DIRECTORY "${extract_1_dir}" + LOGNAME "extract-${TARGET_TRIPLET}-${packstem}" + ) + file(GLOB package_path "${extract_0_dir}/packages/intel.oneapi.${package_infix}.mkl.devel,v=2023.0.0-*") + cmake_path(GET package_path STEM LAST_ONLY packstem) + message(STATUS "Extracting ${packstem}") + vcpkg_execute_required_process( + COMMAND "${CMAKE_COMMAND}" "-E" "tar" "-xf" "${package_path}/cupPayload.cup" + "_installdir/mkl/2023.0.0/bin" + "_installdir/mkl/2023.0.0/include" + "_installdir/mkl/2023.0.0/lib" + WORKING_DIRECTORY "${extract_1_dir}" + LOGNAME "extract-${TARGET_TRIPLET}-${packstem}" + ) + file(GLOB package_path "${extract_0_dir}/packages/intel.oneapi.${package_infix}.openmp,v=2023.0.0-*") + cmake_path(GET package_path STEM LAST_ONLY packstem) + message(STATUS "Extracting ${packstem}") + vcpkg_execute_required_process( + COMMAND "${CMAKE_COMMAND}" "-E" "tar" "-xf" "${package_path}/cupPayload.cup" + "_installdir/compiler/2023.0.0" + WORKING_DIRECTORY "${extract_1_dir}" + LOGNAME "extract-${TARGET_TRIPLET}-${packstem}" + ) + + set(mkl_dir "${extract_1_dir}/_installdir/mkl/2023.0.0") + file(COPY "${mkl_dir}/include/" DESTINATION "${CURRENT_PACKAGES_DIR}/include") + file(COPY "${mkl_dir}/${package_libdir}/" DESTINATION "${CURRENT_PACKAGES_DIR}/lib/intel64") + if(VCPKG_LIBRARY_LINKAGE STREQUAL "dynamic") + set(to_remove_suffix .a) + elseif(VCPKG_TARGET_IS_OSX) + set(to_remove_suffix .dylib) + else() + set(to_remove_suffix .so) + endif() + file(GLOB_RECURSE files_to_remove + "${CURRENT_PACKAGES_DIR}/lib/intel64/*${to_remove_suffix}" + "${CURRENT_PACKAGES_DIR}/lib/intel64/*${to_remove_suffix}.?" + ) + file(REMOVE ${files_to_remove}) + file(COPY_FILE "${mkl_dir}/lib/pkgconfig/${main_pc_file}" "${CURRENT_PACKAGES_DIR}/lib/pkgconfig/${main_pc_file}") + vcpkg_replace_string("${CURRENT_PACKAGES_DIR}/lib/pkgconfig/${main_pc_file}" "\${exec_prefix}/${package_libdir}" "\${exec_prefix}/lib/intel64" IGNORE_UNCHANGED) + + set(compiler_dir "${extract_1_dir}/_installdir/compiler/2023.0.0") + if(threading STREQUAL "intel_thread") + file(COPY "${compiler_dir}/${compiler_libdir}/" DESTINATION "${CURRENT_PACKAGES_DIR}/lib/intel64") + file(COPY_FILE "${compiler_dir}/lib/pkgconfig/openmp.pc" "${CURRENT_PACKAGES_DIR}/lib/pkgconfig/libiomp5.pc") + vcpkg_replace_string("${CURRENT_PACKAGES_DIR}/lib/pkgconfig/libiomp5.pc" "/${compiler_libdir}/" "/lib/intel64/" IGNORE_UNCHANGED) + vcpkg_replace_string("${CURRENT_PACKAGES_DIR}/lib/pkgconfig/${main_pc_file}" "openmp" "libiomp5") + endif() +endif() + +file(COPY_FILE "${CURRENT_PACKAGES_DIR}/lib/pkgconfig/${main_pc_file}" "${CURRENT_PACKAGES_DIR}/lib/pkgconfig/mkl.pc") +if(NOT VCPKG_BUILD_TYPE) + file(MAKE_DIRECTORY "${CURRENT_PACKAGES_DIR}/debug/lib/pkgconfig") + file(GLOB pc_files RELATIVE "${CURRENT_PACKAGES_DIR}/lib/pkgconfig" "${CURRENT_PACKAGES_DIR}/lib/pkgconfig/*.pc") + foreach(file IN LISTS pc_files) + file(COPY_FILE "${CURRENT_PACKAGES_DIR}/lib/pkgconfig/${file}" "${CURRENT_PACKAGES_DIR}/debug/lib/pkgconfig/${file}") + vcpkg_replace_string("${CURRENT_PACKAGES_DIR}/debug/lib/pkgconfig/${file}" "/include" "/../include") + if(NOT VCPKG_TARGET_IS_WINDOWS) + vcpkg_replace_string("${CURRENT_PACKAGES_DIR}/debug/lib/pkgconfig/${file}" "/lib/intel64" "/../lib/intel64") + endif() + endforeach() +endif() + +file(COPY "${mkl_dir}/lib/cmake/" DESTINATION "${CURRENT_PACKAGES_DIR}/share/") +vcpkg_replace_string("${CURRENT_PACKAGES_DIR}/share/mkl/MKLConfig.cmake" "MKL_CMAKE_PATH}/../../../" "MKL_CMAKE_PATH}/../../") +vcpkg_replace_string("${CURRENT_PACKAGES_DIR}/share/mkl/MKLConfig.cmake" "redist/\${MKL_ARCH}" "bin") +if(${VCPKG_LIBRARY_LINKAGE} STREQUAL "static") +vcpkg_replace_string("${CURRENT_PACKAGES_DIR}/share/mkl/MKLConfig.cmake" "define_param(MKL_LINK DEFAULT_MKL_LINK MKL_LINK_LIST)" +[[define_param(MKL_LINK DEFAULT_MKL_LINK MKL_LINK_LIST) + set(MKL_LINK "static") +]]) +endif() +#TODO: Hardcode settings from portfile in config.cmake +#TODO: Give lapack/blas information about the correct BLA_VENDOR depending on settings. + +file(INSTALL "${mkl_dir}/licensing" DESTINATION "${CURRENT_PACKAGES_DIR}/share/${PORT}") +file(GLOB package_path "${extract_0_dir}/packages/intel.oneapi.${package_infix}.mkl.product,v=2023.0.0-*") +vcpkg_install_copyright(FILE_LIST "${package_path}/licenses/license.htm") + +file(REMOVE_RECURSE + "${extract_0_dir}" + "${extract_1_dir}" + "${CURRENT_PACKAGES_DIR}/lib/intel64/cmake" + "${CURRENT_PACKAGES_DIR}/lib/intel64/pkgconfig" +) + +file(INSTALL "${CMAKE_CURRENT_LIST_DIR}/usage" DESTINATION "${CURRENT_PACKAGES_DIR}/share/${PORT}") diff --git a/external_ports/intel-mkl/usage b/external_ports/intel-mkl/usage new file mode 100644 index 00000000..b8ee798f --- /dev/null +++ b/external_ports/intel-mkl/usage @@ -0,0 +1,4 @@ +intel-mkl provides CMake targets: + + find_package(MKL CONFIG REQUIRED) + target_link_libraries(main PRIVATE MKL::MKL) diff --git a/external_ports/intel-mkl/vcpkg.json b/external_ports/intel-mkl/vcpkg.json new file mode 100644 index 00000000..fc0a76ec --- /dev/null +++ b/external_ports/intel-mkl/vcpkg.json @@ -0,0 +1,16 @@ +{ + "name": "intel-mkl", + "version": "2023.0.0", + "port-version": 5, + "description": "Intel® Math Kernel Library (Intel® MKL) accelerates math processing routines, increases application performance, and reduces development time on Intel® processors.", + "homepage": "https://www.intel.com/content/www/us/en/developer/tools/oneapi/onemkl.html", + "license": null, + "supports": "(windows | linux | osx) & x64", + "dependencies": [ + { + "name": "vcpkg-tool-lessmsi", + "host": true, + "platform": "windows" + } + ] +} From 6b7be050e9a820dd9b4e909e0c47cec6849a954f Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Tue, 15 Apr 2025 23:05:38 +0200 Subject: [PATCH 05/56] chore(vcpkg): Switch MKL build to lp64 + sequential --- external_ports/intel-mkl/portfile.cmake | 12 ++++-------- 1 file changed, 4 insertions(+), 8 deletions(-) diff --git a/external_ports/intel-mkl/portfile.cmake b/external_ports/intel-mkl/portfile.cmake index 908a5281..b07c79f1 100644 --- a/external_ports/intel-mkl/portfile.cmake +++ b/external_ports/intel-mkl/portfile.cmake @@ -1,7 +1,7 @@ # This package installs Intel MKL on Linux, macOS and Windows for x64. # Configuration: -# - ilp64 -# - dynamic CRT: intel_thread, static CRT: sequential +# - lp64 +# - sequential set(VCPKG_POLICY_EMPTY_PACKAGE enabled) @@ -44,13 +44,9 @@ vcpkg_download_distfile(installer_path ) # Note: intel_thread and lp64 are the defaults. -set(interface "ilp64") # or ilp64; ilp == 64 bit int api +set(interface "lp64") # or ilp64; ilp == 64 bit int api #https://www.intel.com/content/www/us/en/develop/documentation/onemkl-linux-developer-guide/top/linking-your-application-with-onemkl/linking-in-detail/linking-with-interface-libraries/using-the-ilp64-interface-vs-lp64-interface.html -if(VCPKG_CRT_LINKAGE STREQUAL "dynamic") - set(threading "intel_thread") #sequential or intel_thread or tbb_thread or pgi_thread -else() - set(threading "sequential") -endif() +set(threading "sequential") if(threading STREQUAL "intel_thread") set(short_thread "iomp") else() From 8f935a21e35a5734fed5d906fb728d00e045edcd Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Tue, 15 Apr 2025 23:47:58 +0200 Subject: [PATCH 06/56] fix: Exclude vcpkg *.cmake files from formatting These come from a external project and shouldn't be re-formatted. --- CMakeLists.txt | 2 ++ 1 file changed, 2 insertions(+) diff --git a/CMakeLists.txt b/CMakeLists.txt index 637b7d0c..794a66f9 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -25,6 +25,8 @@ option(GPRAT_ENABLE_FORMAT_TARGETS "Enable clang-format / cmake-format targets" ${PROJECT_IS_TOP_LEVEL}) if(GPRAT_ENABLE_FORMAT_TARGETS) + set(CMAKE_FORMAT_EXCLUDE "^external_ports/") + find_package(format QUIET) if(NOT format_FOUND) include(FetchContent) From 5976c4dcfb31f7adae899ea0b26016142c667eb2 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Fri, 11 Jul 2025 06:42:49 -0400 Subject: [PATCH 07/56] refactor!(core): Move include files into gprat/ subdirectory This is a separate commit for git's rename tracking --- bindings/gprat_py.cpp | 2 +- bindings/utils_py.cpp | 4 ++-- core/include/{ => gprat}/cpu/adapter_cblas_fp32.hpp | 0 core/include/{ => gprat}/cpu/adapter_cblas_fp64.hpp | 0 core/include/{ => gprat}/cpu/gp_algorithms.hpp | 0 core/include/{ => gprat}/cpu/gp_functions.hpp | 0 core/include/{ => gprat}/cpu/gp_optimizer.hpp | 0 core/include/{ => gprat}/cpu/gp_uncertainty.hpp | 0 core/include/{ => gprat}/cpu/tiled_algorithms.hpp | 0 core/include/{ => gprat}/gp_hyperparameters.hpp | 0 core/include/{ => gprat}/gp_kernels.hpp | 0 core/include/{ => gprat}/gprat_c.hpp | 0 core/include/{ => gprat}/gpu/adapter_cublas.cuh | 0 core/include/{ => gprat}/gpu/cuda_kernels.cuh | 0 core/include/{ => gprat}/gpu/cuda_utils.cuh | 0 core/include/{ => gprat}/gpu/gp_algorithms.cuh | 0 core/include/{ => gprat}/gpu/gp_functions.cuh | 0 core/include/{ => gprat}/gpu/gp_optimizer.cuh | 0 core/include/{ => gprat}/gpu/gp_uncertainty.cuh | 0 core/include/{ => gprat}/gpu/tiled_algorithms.cuh | 0 core/include/{ => gprat}/target.hpp | 0 core/include/{ => gprat}/utils_c.hpp | 0 examples/gprat_cpp/src/execute.cpp | 4 ++-- test/src/output_correctness.cpp | 4 ++-- 24 files changed, 7 insertions(+), 7 deletions(-) rename core/include/{ => gprat}/cpu/adapter_cblas_fp32.hpp (100%) rename core/include/{ => gprat}/cpu/adapter_cblas_fp64.hpp (100%) rename core/include/{ => gprat}/cpu/gp_algorithms.hpp (100%) rename core/include/{ => gprat}/cpu/gp_functions.hpp (100%) rename core/include/{ => gprat}/cpu/gp_optimizer.hpp (100%) rename core/include/{ => gprat}/cpu/gp_uncertainty.hpp (100%) rename core/include/{ => gprat}/cpu/tiled_algorithms.hpp (100%) rename core/include/{ => gprat}/gp_hyperparameters.hpp (100%) rename core/include/{ => gprat}/gp_kernels.hpp (100%) rename core/include/{ => gprat}/gprat_c.hpp (100%) rename core/include/{ => gprat}/gpu/adapter_cublas.cuh (100%) rename core/include/{ => gprat}/gpu/cuda_kernels.cuh (100%) rename core/include/{ => gprat}/gpu/cuda_utils.cuh (100%) rename core/include/{ => gprat}/gpu/gp_algorithms.cuh (100%) rename core/include/{ => gprat}/gpu/gp_functions.cuh (100%) rename core/include/{ => gprat}/gpu/gp_optimizer.cuh (100%) rename core/include/{ => gprat}/gpu/gp_uncertainty.cuh (100%) rename core/include/{ => gprat}/gpu/tiled_algorithms.cuh (100%) rename core/include/{ => gprat}/target.hpp (100%) rename core/include/{ => gprat}/utils_c.hpp (100%) diff --git a/bindings/gprat_py.cpp b/bindings/gprat_py.cpp index b18d2279..4d144cff 100644 --- a/bindings/gprat_py.cpp +++ b/bindings/gprat_py.cpp @@ -1,4 +1,4 @@ -#include "gprat_c.hpp" +#include "gprat/gprat_c.hpp" #include #include diff --git a/bindings/utils_py.cpp b/bindings/utils_py.cpp index 277e40ef..5a918c40 100644 --- a/bindings/utils_py.cpp +++ b/bindings/utils_py.cpp @@ -1,5 +1,5 @@ -#include "target.hpp" -#include "utils_c.hpp" +#include "gprat/target.hpp" +#include "gprat/utils_c.hpp" #include #include diff --git a/core/include/cpu/adapter_cblas_fp32.hpp b/core/include/gprat/cpu/adapter_cblas_fp32.hpp similarity index 100% rename from core/include/cpu/adapter_cblas_fp32.hpp rename to core/include/gprat/cpu/adapter_cblas_fp32.hpp diff --git a/core/include/cpu/adapter_cblas_fp64.hpp b/core/include/gprat/cpu/adapter_cblas_fp64.hpp similarity index 100% rename from core/include/cpu/adapter_cblas_fp64.hpp rename to core/include/gprat/cpu/adapter_cblas_fp64.hpp diff --git a/core/include/cpu/gp_algorithms.hpp b/core/include/gprat/cpu/gp_algorithms.hpp similarity index 100% rename from core/include/cpu/gp_algorithms.hpp rename to core/include/gprat/cpu/gp_algorithms.hpp diff --git a/core/include/cpu/gp_functions.hpp b/core/include/gprat/cpu/gp_functions.hpp similarity index 100% rename from core/include/cpu/gp_functions.hpp rename to core/include/gprat/cpu/gp_functions.hpp diff --git a/core/include/cpu/gp_optimizer.hpp b/core/include/gprat/cpu/gp_optimizer.hpp similarity index 100% rename from core/include/cpu/gp_optimizer.hpp rename to core/include/gprat/cpu/gp_optimizer.hpp diff --git a/core/include/cpu/gp_uncertainty.hpp b/core/include/gprat/cpu/gp_uncertainty.hpp similarity index 100% rename from core/include/cpu/gp_uncertainty.hpp rename to core/include/gprat/cpu/gp_uncertainty.hpp diff --git a/core/include/cpu/tiled_algorithms.hpp b/core/include/gprat/cpu/tiled_algorithms.hpp similarity index 100% rename from core/include/cpu/tiled_algorithms.hpp rename to core/include/gprat/cpu/tiled_algorithms.hpp diff --git a/core/include/gp_hyperparameters.hpp b/core/include/gprat/gp_hyperparameters.hpp similarity index 100% rename from core/include/gp_hyperparameters.hpp rename to core/include/gprat/gp_hyperparameters.hpp diff --git a/core/include/gp_kernels.hpp b/core/include/gprat/gp_kernels.hpp similarity index 100% rename from core/include/gp_kernels.hpp rename to core/include/gprat/gp_kernels.hpp diff --git a/core/include/gprat_c.hpp b/core/include/gprat/gprat_c.hpp similarity index 100% rename from core/include/gprat_c.hpp rename to core/include/gprat/gprat_c.hpp diff --git a/core/include/gpu/adapter_cublas.cuh b/core/include/gprat/gpu/adapter_cublas.cuh similarity index 100% rename from core/include/gpu/adapter_cublas.cuh rename to core/include/gprat/gpu/adapter_cublas.cuh diff --git a/core/include/gpu/cuda_kernels.cuh b/core/include/gprat/gpu/cuda_kernels.cuh similarity index 100% rename from core/include/gpu/cuda_kernels.cuh rename to core/include/gprat/gpu/cuda_kernels.cuh diff --git a/core/include/gpu/cuda_utils.cuh b/core/include/gprat/gpu/cuda_utils.cuh similarity index 100% rename from core/include/gpu/cuda_utils.cuh rename to core/include/gprat/gpu/cuda_utils.cuh diff --git a/core/include/gpu/gp_algorithms.cuh b/core/include/gprat/gpu/gp_algorithms.cuh similarity index 100% rename from core/include/gpu/gp_algorithms.cuh rename to core/include/gprat/gpu/gp_algorithms.cuh diff --git a/core/include/gpu/gp_functions.cuh b/core/include/gprat/gpu/gp_functions.cuh similarity index 100% rename from core/include/gpu/gp_functions.cuh rename to core/include/gprat/gpu/gp_functions.cuh diff --git a/core/include/gpu/gp_optimizer.cuh b/core/include/gprat/gpu/gp_optimizer.cuh similarity index 100% rename from core/include/gpu/gp_optimizer.cuh rename to core/include/gprat/gpu/gp_optimizer.cuh diff --git a/core/include/gpu/gp_uncertainty.cuh b/core/include/gprat/gpu/gp_uncertainty.cuh similarity index 100% rename from core/include/gpu/gp_uncertainty.cuh rename to core/include/gprat/gpu/gp_uncertainty.cuh diff --git a/core/include/gpu/tiled_algorithms.cuh b/core/include/gprat/gpu/tiled_algorithms.cuh similarity index 100% rename from core/include/gpu/tiled_algorithms.cuh rename to core/include/gprat/gpu/tiled_algorithms.cuh diff --git a/core/include/target.hpp b/core/include/gprat/target.hpp similarity index 100% rename from core/include/target.hpp rename to core/include/gprat/target.hpp diff --git a/core/include/utils_c.hpp b/core/include/gprat/utils_c.hpp similarity index 100% rename from core/include/utils_c.hpp rename to core/include/gprat/utils_c.hpp diff --git a/examples/gprat_cpp/src/execute.cpp b/examples/gprat_cpp/src/execute.cpp index 8c415727..fa36357a 100644 --- a/examples/gprat_cpp/src/execute.cpp +++ b/examples/gprat_cpp/src/execute.cpp @@ -1,5 +1,5 @@ -#include "gprat_c.hpp" -#include "utils_c.hpp" +#include "gprat/gprat_c.hpp" +#include "gprat/utils_c.hpp" #include #include #include diff --git a/test/src/output_correctness.cpp b/test/src/output_correctness.cpp index 1fc73536..1c61fdd7 100644 --- a/test/src/output_correctness.cpp +++ b/test/src/output_correctness.cpp @@ -1,5 +1,5 @@ -#include "gprat_c.hpp" -#include "utils_c.hpp" +#include "gprat/gprat_c.hpp" +#include "gprat/utils_c.hpp" #include #include From 0017764ded01e76bab3c0ef52f03803635cbc43e Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Mon, 14 Jul 2025 11:00:44 -0400 Subject: [PATCH 08/56] refactor!(core): Move everything into the gprat namespace --- bindings/gprat_py.cpp | 15 ++-- bindings/utils_py.cpp | 19 +++-- core/include/gprat/cpu/adapter_cblas_fp32.hpp | 15 +++- core/include/gprat/cpu/adapter_cblas_fp64.hpp | 15 +++- core/include/gprat/cpu/gp_algorithms.hpp | 26 ++++-- core/include/gprat/cpu/gp_functions.hpp | 36 ++++---- core/include/gprat/cpu/gp_optimizer.hpp | 41 +++++----- core/include/gprat/cpu/gp_uncertainty.hpp | 12 ++- core/include/gprat/cpu/tiled_algorithms.hpp | 22 +++-- core/include/gprat/detail/config.hpp | 26 ++++++ core/include/gprat/gp_hyperparameters.hpp | 15 ++-- core/include/gprat/gp_kernels.hpp | 18 ++-- core/include/gprat/gprat_c.hpp | 29 ++++--- core/include/gprat/gpu/adapter_cublas.cuh | 20 +++-- core/include/gprat/gpu/cuda_kernels.cuh | 14 +++- core/include/gprat/gpu/cuda_utils.cuh | 23 ++++-- core/include/gprat/gpu/gp_algorithms.cuh | 64 ++++++++------- core/include/gprat/gpu/gp_functions.cuh | 46 ++++++----- core/include/gprat/gpu/gp_optimizer.cuh | 43 ++++++---- core/include/gprat/gpu/gp_uncertainty.cuh | 20 +++-- core/include/gprat/gpu/tiled_algorithms.cuh | 65 ++++++++------- core/include/gprat/target.hpp | 15 ++-- core/include/gprat/utils_c.hpp | 14 ++-- core/src/cpu/adapter_cblas_fp32.cpp | 6 +- core/src/cpu/adapter_cblas_fp64.cpp | 6 +- core/src/cpu/gp_algorithms.cpp | 16 ++-- core/src/cpu/gp_functions.cpp | 31 ++++--- core/src/cpu/gp_optimizer.cpp | 32 ++++---- core/src/cpu/gp_uncertainty.cpp | 6 +- core/src/cpu/tiled_algorithms.cpp | 19 +++-- core/src/gp_hyperparameters.cpp | 7 +- core/src/gp_kernels.cpp | 17 ++-- core/src/gprat_c.cpp | 31 ++++--- core/src/gpu/adapter_cublas.cu | 6 +- core/src/gpu/cuda_kernels.cu | 8 +- core/src/gpu/gp_algorithms.cu | 82 ++++++++++--------- core/src/gpu/gp_functions.cu | 49 ++++++----- core/src/gpu/gp_optimizer.cu | 40 ++++----- core/src/gpu/gp_uncertainty.cu | 17 ++-- core/src/gpu/tiled_algorithms.cu | 63 +++++++------- core/src/target.cpp | 9 +- core/src/utils_c.cpp | 7 +- examples/gprat_cpp/src/execute.cpp | 15 ++-- test/src/output_correctness.cpp | 21 ++--- 44 files changed, 653 insertions(+), 448 deletions(-) create mode 100644 core/include/gprat/detail/config.hpp diff --git a/bindings/gprat_py.cpp b/bindings/gprat_py.cpp index 4d144cff..b122df75 100644 --- a/bindings/gprat_py.cpp +++ b/bindings/gprat_py.cpp @@ -1,4 +1,5 @@ #include "gprat/gprat_c.hpp" + #include #include @@ -31,19 +32,19 @@ void init_gprat(py::module &m) // Set hyperparameters to default values in `AdamParams` class, unless // specified. Python object has full access to each hyperparameter and a // string representation `__repr__`. - py::class_(m, "AdamParams") + py::class_(m, "AdamParams") .def(py::init(), py::arg("learning_rate") = 0.001, py::arg("beta1") = 0.9, py::arg("beta2") = 0.999, py::arg("epsilon") = 1e-8, py::arg("opt_iter") = 0) - .def_readwrite("learning_rate", &gprat_hyper::AdamParams::learning_rate) - .def_readwrite("beta1", &gprat_hyper::AdamParams::beta1) - .def_readwrite("beta2", &gprat_hyper::AdamParams::beta2) - .def_readwrite("epsilon", &gprat_hyper::AdamParams::epsilon) - .def_readwrite("opt_iter", &gprat_hyper::AdamParams::opt_iter) - .def("__repr__", &gprat_hyper::AdamParams::repr); + .def_readwrite("learning_rate", &gprat::AdamParams::learning_rate) + .def_readwrite("beta1", &gprat::AdamParams::beta1) + .def_readwrite("beta2", &gprat::AdamParams::beta2) + .def_readwrite("epsilon", &gprat::AdamParams::epsilon) + .def_readwrite("opt_iter", &gprat::AdamParams::opt_iter) + .def("__repr__", &gprat::AdamParams::repr); // Initializes Gaussian Process with `GP` class. Sets default parameters for // squared exponential kernel, number of regressors and trainable, unless diff --git a/bindings/utils_py.cpp b/bindings/utils_py.cpp index 5a918c40..0fc35506 100644 --- a/bindings/utils_py.cpp +++ b/bindings/utils_py.cpp @@ -1,5 +1,6 @@ #include "gprat/target.hpp" #include "gprat/utils_c.hpp" + #include #include @@ -32,7 +33,7 @@ void start_hpx_wrapper(std::vector args, std::size_t n_cores) } argv.push_back(nullptr); int argc = static_cast(args.size()); - utils::start_hpx_runtime(argc, argv.data()); + gprat::start_hpx_runtime(argc, argv.data()); } /** @@ -43,7 +44,7 @@ void start_hpx_wrapper(std::vector args, std::size_t n_cores) void init_utils(py::module &m) { m.def("compute_train_tiles", - &utils::compute_train_tiles, + &gprat::compute_train_tiles, py::arg("n_samples"), py::arg("n_tile_size"), R"pbdoc( @@ -58,7 +59,7 @@ void init_utils(py::module &m) )pbdoc"); m.def("compute_train_tile_size", - &utils::compute_train_tile_size, + &gprat::compute_train_tile_size, py::arg("n_samples"), py::arg("n_tiles"), R"pbdoc( @@ -73,7 +74,7 @@ void init_utils(py::module &m) )pbdoc"); m.def("compute_test_tiles", - &utils::compute_test_tiles, + &gprat::compute_test_tiles, py::arg("m_samples"), py::arg("n_tiles"), py::arg("n_tile_size"), @@ -90,7 +91,7 @@ void init_utils(py::module &m) )pbdoc"); m.def("print_vector", - &utils::print_vector, + &gprat::print_vector, py::arg("vec"), py::arg("start") = 0, py::arg("end") = -1, @@ -98,11 +99,11 @@ void init_utils(py::module &m) "Print elements of a vector with optional start, end, and separator parameters"); m.def("start_hpx", &start_hpx_wrapper, py::arg("args"), py::arg("n_cores")); // Using the wrapper function - m.def("resume_hpx", &utils::resume_hpx_runtime); - m.def("suspend_hpx", &utils::suspend_hpx_runtime); - m.def("stop_hpx", &utils::stop_hpx_runtime); + m.def("resume_hpx", &gprat::resume_hpx_runtime); + m.def("suspend_hpx", &gprat::suspend_hpx_runtime); + m.def("stop_hpx", &gprat::stop_hpx_runtime); - m.def("compiled_with_cuda", &utils::compiled_with_cuda, "Check if the code was compiled with CUDA support"); + m.def("compiled_with_cuda", &gprat::compiled_with_cuda, "Check if the code was compiled with CUDA support"); m.def("print_available_gpus", &gprat::print_available_gpus, "Print available GPUs with their properties"); m.def("gpu_count", &gprat::gpu_count, "Return the number of available GPUs"); diff --git a/core/include/gprat/cpu/adapter_cblas_fp32.hpp b/core/include/gprat/cpu/adapter_cblas_fp32.hpp index 9cf21915..fa6272b5 100644 --- a/core/include/gprat/cpu/adapter_cblas_fp32.hpp +++ b/core/include/gprat/cpu/adapter_cblas_fp32.hpp @@ -1,8 +1,15 @@ -#ifndef CPU_ADAPTER_CBLAS_FP32_H -#define CPU_ADAPTER_CBLAS_FP32_H +#ifndef GPRAT_CPU_ADAPTER_CBLAS_FP32_HPP +#define GPRAT_CPU_ADAPTER_CBLAS_FP32_HPP + +#pragma once + +#include "gprat/detail/config.hpp" #include #include + +GPRAT_NS_BEGIN + using vector_future = hpx::shared_future>; // Constants that are compatible with CBLAS @@ -145,4 +152,6 @@ vector_future axpy(vector_future f_y, vector_future f_x, const int N); */ float dot(std::vector a, std::vector b, const int N); -#endif // end of CPU_ADAPTER_CBLAS_FP32_H +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/cpu/adapter_cblas_fp64.hpp b/core/include/gprat/cpu/adapter_cblas_fp64.hpp index b3c95420..fbf81b9d 100644 --- a/core/include/gprat/cpu/adapter_cblas_fp64.hpp +++ b/core/include/gprat/cpu/adapter_cblas_fp64.hpp @@ -1,13 +1,18 @@ -#ifndef CPU_ADAPTER_CBLAS_FP64_H -#define CPU_ADAPTER_CBLAS_FP64_H +#ifndef GPRAT_CPU_ADAPTER_CBLAS_FP64_HPP +#define GPRAT_CPU_ADAPTER_CBLAS_FP64_HPP + +#pragma once + +#include "gprat/detail/config.hpp" #include #include +GPRAT_NS_BEGIN + using vector_future = hpx::shared_future>; // Constants that are compatible with CBLAS - typedef enum BLAS_TRANSPOSE { Blas_no_trans = 111, Blas_trans = 112 } BLAS_TRANSPOSE; typedef enum BLAS_SIDE { Blas_left = 141, Blas_right = 142 } BLAS_SIDE; @@ -147,4 +152,6 @@ vector_future axpy(vector_future f_y, vector_future f_x, const int N); */ double dot(std::vector a, std::vector b, const int N); -#endif // end of CPU_ADAPTER_CBLAS_FP64_H +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/cpu/gp_algorithms.hpp b/core/include/gprat/cpu/gp_algorithms.hpp index b8a6f043..2ad66542 100644 --- a/core/include/gprat/cpu/gp_algorithms.hpp +++ b/core/include/gprat/cpu/gp_algorithms.hpp @@ -1,9 +1,15 @@ -#ifndef CPU_GP_ALGORITHMS_H -#define CPU_GP_ALGORITHMS_H +#ifndef GPRAT_CPU_GP_ALGORITHMS_HPP +#define GPRAT_CPU_GP_ALGORITHMS_HPP + +#pragma once + +#include "gprat/detail/config.hpp" +#include "gprat/gp_kernels.hpp" -#include "gp_kernels.hpp" #include +GPRAT_NS_BEGIN + namespace cpu { @@ -22,7 +28,7 @@ namespace cpu double compute_covariance_function(std::size_t i_global, std::size_t j_global, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, const std::vector &i_input, const std::vector &j_input); @@ -44,7 +50,7 @@ std::vector gen_tile_covariance( std::size_t col, std::size_t N, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, const std::vector &input); /** @@ -66,7 +72,7 @@ std::vector gen_tile_full_prior_covariance( std::size_t col, std::size_t N, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, const std::vector &input); /** @@ -88,7 +94,7 @@ std::vector gen_tile_prior_covariance( std::size_t col, std::size_t N, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, const std::vector &input); /** @@ -111,7 +117,7 @@ std::vector gen_tile_cross_covariance( std::size_t N_row, std::size_t N_col, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, const std::vector &row_input, const std::vector &col_input); @@ -170,4 +176,6 @@ std::vector gen_tile_identity(std::size_t N); } // end of namespace cpu -#endif // end of CPU_GP_ALGORITHMS_H +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/cpu/gp_functions.hpp b/core/include/gprat/cpu/gp_functions.hpp index 7079bab6..fcd41996 100644 --- a/core/include/gprat/cpu/gp_functions.hpp +++ b/core/include/gprat/cpu/gp_functions.hpp @@ -1,10 +1,16 @@ -#ifndef CPU_GP_FUNCTIONS_H -#define CPU_GP_FUNCTIONS_H +#ifndef GPRAT_CPU_GP_FUNCTIONS_HPP +#define GPRAT_CPU_GP_FUNCTIONS_HPP + +#pragma once + +#include "gprat/detail/config.hpp" +#include "gprat/gp_hyperparameters.hpp" +#include "gprat/gp_kernels.hpp" -#include "gp_hyperparameters.hpp" -#include "gp_kernels.hpp" #include +GPRAT_NS_BEGIN + namespace cpu { @@ -22,7 +28,7 @@ namespace cpu */ std::vector> cholesky(const std::vector &training_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int n_regressors); @@ -46,7 +52,7 @@ std::vector predict(const std::vector &training_input, const std::vector &training_output, const std::vector &test_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int m_tiles, @@ -72,7 +78,7 @@ std::vector> predict_with_uncertainty( const std::vector &training_input, const std::vector &training_output, const std::vector &test_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int m_tiles, @@ -98,7 +104,7 @@ std::vector> predict_with_full_cov( const std::vector &training_input, const std::vector &training_output, const std::vector &test_data, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int m_tiles, @@ -119,7 +125,7 @@ std::vector> predict_with_full_cov( */ double compute_loss(const std::vector &training_input, const std::vector &training_output, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int n_regressors); @@ -146,8 +152,8 @@ optimize(const std::vector &training_input, int n_tiles, int n_tile_size, int n_regressors, - const gprat_hyper::AdamParams &adam_params, - gprat_hyper::SEKParams &sek_params, + const AdamParams &adam_params, + SEKParams &sek_params, std::vector trainable_params); /** @@ -173,11 +179,13 @@ double optimize_step(const std::vector &training_input, int n_tiles, int n_tile_size, int n_regressors, - gprat_hyper::AdamParams &adam_params, - gprat_hyper::SEKParams &sek_params, + AdamParams &adam_params, + SEKParams &sek_params, std::vector trainable_params, int iter); } // end of namespace cpu -#endif // end of CPU_GP_FUNCTIONS_H +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/cpu/gp_optimizer.hpp b/core/include/gprat/cpu/gp_optimizer.hpp index c632e87b..ff9dc1b7 100644 --- a/core/include/gprat/cpu/gp_optimizer.hpp +++ b/core/include/gprat/cpu/gp_optimizer.hpp @@ -1,10 +1,16 @@ -#ifndef CPU_GP_OPTIMIZER_H -#define CPU_GP_OPTIMIZER_H +#ifndef GPRAT_CPU_GP_OPTIMIZER_H +#define GPRAT_CPU_GP_OPTIMIZER_H + +#pragma once + +#include "gprat/detail/config.hpp" +#include "gprat/gp_hyperparameters.hpp" +#include "gprat/gp_kernels.hpp" -#include "gp_hyperparameters.hpp" -#include "gp_kernels.hpp" #include +GPRAT_NS_BEGIN + namespace cpu { @@ -54,7 +60,7 @@ double compute_sigmoid(double parameter); double compute_covariance_distance(std::size_t i_global, std::size_t j_global, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, const std::vector &i_input, const std::vector &j_input); @@ -75,7 +81,7 @@ std::vector gen_tile_distance( std::size_t col, std::size_t N, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, const std::vector &input); /** @@ -90,11 +96,7 @@ std::vector gen_tile_distance( * @return A quadratic tile of the covariance matrix of size N x N */ std::vector gen_tile_covariance_with_distance( - std::size_t row, - std::size_t col, - std::size_t N, - const gprat_hyper::SEKParams &sek_params, - const std::vector &distance); + std::size_t row, std::size_t col, std::size_t N, const SEKParams &sek_params, const std::vector &distance); /** * @brief Generate a derivative tile w.r.t. vertical_lengthscale v @@ -105,8 +107,7 @@ std::vector gen_tile_covariance_with_distance( * * @return A quadratic tile of the derivative of v of size N x N */ -std::vector -gen_tile_grad_v(std::size_t N, const gprat_hyper::SEKParams &sek_params, const std::vector &distance); +std::vector gen_tile_grad_v(std::size_t N, const SEKParams &sek_params, const std::vector &distance); /** * @brief Generate a derivative tile w.r.t. lengthscale l @@ -117,8 +118,7 @@ gen_tile_grad_v(std::size_t N, const gprat_hyper::SEKParams &sek_params, const s * * @return A quadratic tile of the derivative of l of size N x N */ -std::vector -gen_tile_grad_l(std::size_t N, const gprat_hyper::SEKParams &sek_params, const std::vector &distance); +std::vector gen_tile_grad_l(std::size_t N, const SEKParams &sek_params, const std::vector &distance); /** * @brief Update biased first raw moment estimate: m_T+1 = beta_1 * m_T + (1 - beta_1) * g_T. @@ -153,11 +153,8 @@ double update_second_moment(double gradient, double v_T, double beta_2); * * @return The updated hyperparameter */ -double adam_step(const double unconstrained_hyperparam, - const gprat_hyper::AdamParams &adam_params, - double m_T, - double v_T, - std::size_t iter); +double adam_step( + const double unconstrained_hyperparam, const AdamParams &adam_params, double m_T, double v_T, std::size_t iter); /** * @brief Compute negative-log likelihood on one tile. @@ -230,4 +227,6 @@ double compute_trace_diag(const std::vector &tile, double trace, std::si } // end of namespace cpu -#endif // end of CPU_GP_OPTIMIZER_H +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/cpu/gp_uncertainty.hpp b/core/include/gprat/cpu/gp_uncertainty.hpp index 28089584..705f7798 100644 --- a/core/include/gprat/cpu/gp_uncertainty.hpp +++ b/core/include/gprat/cpu/gp_uncertainty.hpp @@ -1,9 +1,15 @@ -#ifndef CPU_GP_UNCERTAINTY_H -#define CPU_GP_UNCERTAINTY_H +#ifndef GPRAT_CPU_GP_UNCERTAINTY_HPP +#define GPRAT_CPU_GP_UNCERTAINTY_HPP + +#pragma once + +#include "gprat/detail/config.hpp" #include #include +GPRAT_NS_BEGIN + namespace cpu { @@ -20,4 +26,6 @@ hpx::shared_future> get_matrix_diagonal(hpx::shared_future +GPRAT_NS_BEGIN + using Tiled_matrix = std::vector>>; using Tiled_vector = std::vector>>; @@ -171,8 +177,8 @@ void update_hyperparameter_tiled( const Tiled_matrix &ft_invK, const Tiled_matrix &ft_gradK_param, const Tiled_vector &ft_alpha, - const gprat_hyper::AdamParams &adam_params, - gprat_hyper::SEKParams &sek_params, + const AdamParams &adam_params, + SEKParams &sek_params, int N, std::size_t n_tiles, std::size_t iter, @@ -180,4 +186,6 @@ void update_hyperparameter_tiled( } // end of namespace cpu -#endif // end of CPU_TILED_ALGORITHMS_H +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/detail/config.hpp b/core/include/gprat/detail/config.hpp new file mode 100644 index 00000000..e47a2de7 --- /dev/null +++ b/core/include/gprat/detail/config.hpp @@ -0,0 +1,26 @@ +#ifndef GPRAT_DETAIL_CONFIG_HPP +#define GPRAT_DETAIL_CONFIG_HPP + +#pragma once + +// clang-format off +#define GPRAT_NS gprat::v1 +#define GPRAT_NS_BEGIN namespace gprat { inline namespace v1 { +#define GPRAT_NS_END } } +// clang-format on + +#if defined(_MSC_VER) || defined(__BORLANDC__) || defined(__CODEGEARC__) +#if defined(GPRAT_DYN_LINK) +#if defined(GPRAT_SOURCE) +#define GPRAT_DECL __declspec(dllexport) +#else +#define GPRAT_DECL __declspec(dllimport) +#endif +#endif +#endif + +#if !defined(GPRAT_DECL) +#define GPRAT_DECL +#endif + +#endif diff --git a/core/include/gprat/gp_hyperparameters.hpp b/core/include/gprat/gp_hyperparameters.hpp index cd9cf5a8..9ede4756 100644 --- a/core/include/gprat/gp_hyperparameters.hpp +++ b/core/include/gprat/gp_hyperparameters.hpp @@ -1,10 +1,13 @@ -#ifndef GP_HYPERPARAMETERS_H -#define GP_HYPERPARAMETERS_H +#ifndef GPRAT_GPHYPERPARAMETERS_HPP +#define GPRAT_GPHYPERPARAMETERS_HPP + +#pragma once + +#include "gprat/detail/config.hpp" #include -namespace gprat_hyper -{ +GPRAT_NS_BEGIN /** * @brief Hyperparameters for the Adam optimizer @@ -55,6 +58,6 @@ struct AdamParams std::string repr() const; }; -} // namespace gprat_hyper +GPRAT_NS_END -#endif // GP_HYPERPARAMETERS_H +#endif diff --git a/core/include/gprat/gp_kernels.hpp b/core/include/gprat/gp_kernels.hpp index c1346f32..8b5dc9c1 100644 --- a/core/include/gprat/gp_kernels.hpp +++ b/core/include/gprat/gp_kernels.hpp @@ -1,12 +1,14 @@ -#ifndef GP_KERNELS_H -#define GP_KERNELS_H +#ifndef GPRAT_GPKERNELS_HPP +#define GPRAT_GPKERNELS_HPP -#include +#pragma once -// #include +#include "gprat/detail/config.hpp" -namespace gprat_hyper -{ +#include +#include + +GPRAT_NS_BEGIN /** * @brief Squared Exponential Kernel Parameters @@ -77,6 +79,6 @@ struct SEKParams const double &get_param(std::size_t index) const; }; -} // namespace gprat_hyper +GPRAT_NS_END -#endif // end of GP_KERNELS_H +#endif diff --git a/core/include/gprat/gprat_c.hpp b/core/include/gprat/gprat_c.hpp index 6781d286..2596eaed 100644 --- a/core/include/gprat/gprat_c.hpp +++ b/core/include/gprat/gprat_c.hpp @@ -1,16 +1,18 @@ -#ifndef GPRAT_C_H -#define GPRAT_C_H +#ifndef GPRAT_C_HPP +#define GPRAT_C_HPP + +#pragma once + +#include "gprat/detail/config.hpp" +#include "gprat/gp_hyperparameters.hpp" +#include "gprat/gp_kernels.hpp" +#include "gprat/target.hpp" -#include "gp_hyperparameters.hpp" -#include "gp_kernels.hpp" -#include "target.hpp" #include #include #include -// namespace for GPRat library entities -namespace gprat -{ +GPRAT_NS_BEGIN /** * @brief Data structure for Gaussian Process data @@ -84,7 +86,7 @@ class GP /** * @brief Hyperarameters of the squared exponential kernel */ - gprat_hyper::SEKParams kernel_params; + SEKParams kernel_params; /** * @brief Constructs a Gaussian Process (GP) @@ -203,7 +205,7 @@ class GP * * @return losses */ - std::vector optimize(const gprat_hyper::AdamParams &adam_params); + std::vector optimize(const AdamParams &adam_params); /** * @brief Perform a single optimization step @@ -214,7 +216,7 @@ class GP * * @return loss */ - double optimize_step(gprat_hyper::AdamParams &adam_params, int iter); + double optimize_step(AdamParams &adam_params, int iter); /** * @brief Calculate loss for given data and Gaussian process model @@ -226,6 +228,7 @@ class GP */ std::vector> cholesky(); }; -} // namespace gprat -#endif // end of GPRAT_C_H +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/gpu/adapter_cublas.cuh b/core/include/gprat/gpu/adapter_cublas.cuh index 1a69cb58..05972b36 100644 --- a/core/include/gprat/gpu/adapter_cublas.cuh +++ b/core/include/gprat/gpu/adapter_cublas.cuh @@ -1,10 +1,18 @@ -#ifndef ADAPTER_CUBLAS_H -#define ADAPTER_CUBLAS_H +#ifndef GRRAT_GPU_ADAPTER_CUBLAS_HPP +#define GPRAT_GPU_ADAPTER_CUBLAS_HPP + +#pragma once + +#include "gprat/detail/config.hpp" + +#include "gprat/target.hpp" -#include #include #include -#include + +#include + +GPRAT_NS_BEGIN // Constants, compatible with cuBLAS @@ -262,4 +270,6 @@ inline cublasSideMode_t opposite(cublasSideMode_t side) return (side == CUBLAS_SIDE_LEFT) ? CUBLAS_SIDE_RIGHT : CUBLAS_SIDE_LEFT; } -#endif // end of ADAPTER_CUBLAS_H +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/gpu/cuda_kernels.cuh b/core/include/gprat/gpu/cuda_kernels.cuh index 4daef473..69a48d8f 100644 --- a/core/include/gprat/gpu/cuda_kernels.cuh +++ b/core/include/gprat/gpu/cuda_kernels.cuh @@ -1,5 +1,11 @@ -#ifndef CUDA_KERNELS_H -#define CUDA_KERNELS_H +#ifndef GPRAT_CUDA_KERNELS_HPP +#define GPRAT_CUDA_KERNELS_HPP + +#pragma once + +#include "gprat/detail/config.hpp" + +GPRAT_NS_BEGIN /** * @brief Kernel to transpose a matrix. @@ -11,4 +17,6 @@ */ __global__ void transpose(double *transposed, double *original, std::size_t width, std::size_t height); -#endif // CUDA_KERNELS_H +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/gpu/cuda_utils.cuh b/core/include/gprat/gpu/cuda_utils.cuh index 0c51ea76..029b248c 100644 --- a/core/include/gprat/gpu/cuda_utils.cuh +++ b/core/include/gprat/gpu/cuda_utils.cuh @@ -1,17 +1,21 @@ -#ifndef CUDA_UTILS_H -#define CUDA_UTILS_H +#ifndef GPRAT_CUDA_UTILS_HPP +#define GPRAT_CUDA_UTILS_HPP + +#pragma once + +#include "gprat/detail/config.hpp" +#include "gprat/target.hpp" #include #include #include #include #include -#include #include -#define BLOCK_SIZE 16 +GPRAT_NS_BEGIN -using hpx::cuda::experimental::check_cuda_error; +#define BLOCK_SIZE 16 /** * @brief Copies a vector from the host to the device using the next CUDA stream @@ -25,8 +29,9 @@ using hpx::cuda::experimental::check_cuda_error; * * @return A pointer to the copied vector on the device */ -inline double *copy_to_device(const std::vector &h_vector, gprat::CUDA_GPU &gpu) +inline double *copy_to_device(const std::vector &h_vector, CUDA_GPU &gpu) { + using hpx::cuda::experimental::check_cuda_error; double *d_vector; check_cuda_error(cudaMalloc(&d_vector, h_vector.size() * sizeof(double))); cudaStream_t stream = gpu.next_stream(); @@ -41,6 +46,7 @@ inline double *copy_to_device(const std::vector &h_vector, gprat::CUDA_G */ inline cusolverDnHandle_t create_cusolver_handle() { + using hpx::cuda::experimental::check_cuda_error; cusolverDnHandle_t handle; cusolverDnCreate(&handle); return handle; @@ -60,10 +66,13 @@ inline void destroy(cusolverDnHandle_t handle) { cusolverDnDestroy(handle); } */ inline void free(std::vector> &vector) { + using hpx::cuda::experimental::check_cuda_error; for (auto &ptr : vector) { check_cuda_error(cudaFree(ptr.get())); } } -#endif // end of CUDA_UTILS_H +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/gpu/gp_algorithms.cuh b/core/include/gprat/gpu/gp_algorithms.cuh index 51cbc355..73981b96 100644 --- a/core/include/gprat/gpu/gp_algorithms.cuh +++ b/core/include/gprat/gpu/gp_algorithms.cuh @@ -1,11 +1,17 @@ -#ifndef GPU_GP_ALGORITHMS_H -#define GPU_GP_ALGORITHMS_H +#ifndef GPRAT_GPU_GP_ALGORITHMS_HPP +#define GPRAT_GPU_GP_ALGORITHMS_HPP -#include "gp_kernels.hpp" -#include "target.hpp" +#pragma once + +#include "gprat/detail/config.hpp" + +#include "gprat/gp_kernels.hpp" +#include "gprat/target.hpp" #include #include +GPRAT_NS_BEGIN + namespace gpu { @@ -28,8 +34,8 @@ double *gen_tile_covariance(const double *d_input, const std::size_t tile_column, const std::size_t n_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu); + const SEKParams sek_params, + CUDA_GPU &gpu); /** * @brief Generate the diagonal of a diagonal tile in the prior covariance matrix @@ -51,8 +57,8 @@ double *gen_tile_prior_covariance( const std::size_t tile_column, const std::size_t n_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu); + const SEKParams sek_params, + CUDA_GPU &gpu); /** * @brief Generate a tile of the cross-covariance matrix @@ -77,8 +83,8 @@ double *gen_tile_cross_covariance( const std::size_t n_row_tile_size, const std::size_t n_column_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu); + const SEKParams sek_params, + CUDA_GPU &gpu); /** * @brief Transpose a tile of size n_row_tile_size x n_column_tile_size @@ -92,7 +98,7 @@ double *gen_tile_cross_covariance( hpx::shared_future gen_tile_transpose(std::size_t n_row_tile_size, std::size_t n_column_tile_size, const hpx::shared_future f_tile, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Generate a tile of the output data @@ -104,7 +110,7 @@ hpx::shared_future gen_tile_transpose(std::size_t n_row_tile_size, * @return A tile of the output data of size n_tile_size */ double * -gen_tile_output(const std::size_t row, const std::size_t n_tile_size, const double *d_output, gprat::CUDA_GPU &gpu); +gen_tile_output(const std::size_t row, const std::size_t n_tile_size, const double *d_output, CUDA_GPU &gpu); /** * @brief Compute the L2-error norm over all tiles and elements @@ -126,7 +132,7 @@ double compute_error_norm(const std::size_t n_tiles, * * @return A tile filled with zeros of size N */ -double *gen_tile_zeros(std::size_t n_tile_size, gprat::CUDA_GPU &gpu); +double *gen_tile_zeros(std::size_t n_tile_size, CUDA_GPU &gpu); /** * @brief Allocates the tiled covariance matrix on the device given the training @@ -144,8 +150,8 @@ std::vector> assemble_tiled_covariance_matrix( const std::size_t n_tiles, const std::size_t n_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu); + const SEKParams sek_params, + CUDA_GPU &gpu); /** * @brief Allocates the tiled alpha vector on the device given the training @@ -159,7 +165,7 @@ std::vector> assemble_tiled_covariance_matrix( * @return A tiled alpha vector of size n_tiles x n_tile_size */ std::vector> assemble_alpha_tiles( - const double *d_output, const std::size_t n_tiles, const std::size_t n_tile_size, gprat::CUDA_GPU &gpu); + const double *d_output, const std::size_t n_tiles, const std::size_t n_tile_size, CUDA_GPU &gpu); /** * @brief Allocates the tiled cross covariance matrix on the device given the @@ -185,8 +191,8 @@ std::vector> assemble_cross_covariance_tiles( const std::size_t m_tile_size, const std::size_t n_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu); + const SEKParams sek_params, + CUDA_GPU &gpu); /** * @brief Allocates a tiled vector on the device and initializes it with zeros. @@ -198,7 +204,7 @@ std::vector> assemble_cross_covariance_tiles( * @return A tiled vector of size n_tiles x n_tile_size with zeros */ std::vector> -assemble_tiles_with_zeros(std::size_t n_tile_size, std::size_t n_tiles, gprat::CUDA_GPU &gpu); +assemble_tiles_with_zeros(std::size_t n_tile_size, std::size_t n_tiles, CUDA_GPU &gpu); /** * @brief Allocates the tiled prior covariance matrix on the device given the @@ -218,8 +224,8 @@ std::vector> assemble_prior_K_tiles( const std::size_t m_tiles, const std::size_t m_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu); + const SEKParams sek_params, + CUDA_GPU &gpu); /** * @brief Allocates the posterior covariance matrix. @@ -238,8 +244,8 @@ std::vector> assemble_prior_K_tiles_full( const std::size_t m_tiles, const std::size_t m_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu); + const SEKParams sek_params, + CUDA_GPU &gpu); /** * @brief Allocates the tiled transpose cross covariance matrix on the device @@ -261,7 +267,7 @@ std::vector> assemble_t_cross_covariance_tiles( const std::size_t m_tiles, const std::size_t n_tile_size, const std::size_t m_tile_size, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Allocates the output vector on the device given the training output @@ -272,7 +278,7 @@ std::vector> assemble_t_cross_covariance_tiles( * @param gpu GPU target for computations */ std::vector> assemble_y_tiles( - const double *d_training_output, const std::size_t n_tiles, const std::size_t n_tile_size, gprat::CUDA_GPU &gpu); + const double *d_training_output, const std::size_t n_tiles, const std::size_t n_tile_size, CUDA_GPU &gpu); /** * @brief Allocates the tiled covariance matrix on the device given the training @@ -286,7 +292,7 @@ std::vector> assemble_y_tiles( std::vector copy_tiled_vector_to_host_vector(std::vector> &d_tiles, std::size_t n_tile_size, std::size_t n_tiles, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Moves lower triangular tiles of the covariance matrix to the host. @@ -302,7 +308,7 @@ std::vector> move_lower_tiled_matrix_to_host( const std::vector> &d_tiles, const std::size_t n_tile_size, const std::size_t n_tiles, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Frees the device memory of the lower triangular tiles of the covariance matrix. @@ -314,4 +320,6 @@ void free_lower_tiled_matrix(const std::vector> &d_ } // end of namespace gpu -#endif // end of GPU_GP_ALGORITHMS_H +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/gpu/gp_functions.cuh b/core/include/gprat/gpu/gp_functions.cuh index 6ea5bd0a..f4949def 100644 --- a/core/include/gprat/gpu/gp_functions.cuh +++ b/core/include/gprat/gpu/gp_functions.cuh @@ -1,9 +1,15 @@ #ifndef GPU_GP_FUNCTIONS_H #define GPU_GP_FUNCTIONS_H -#include "gp_hyperparameters.hpp" -#include "gp_kernels.hpp" -#include "target.hpp" +#pragma once + +#include "gprat/detail/config.hpp" + +#include "gprat/gp_hyperparameters.hpp" +#include "gprat/gp_kernels.hpp" +#include "gprat/target.hpp" + +GPRAT_NS_BEGIN namespace gpu { @@ -28,13 +34,13 @@ std::vector predict(const std::vector &training_input, const std::vector &training_output, const std::vector &test_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int m_tiles, int m_tile_size, int n_regressors, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Compute the predictions with uncertainties. @@ -56,13 +62,13 @@ std::vector> predict_with_uncertainty( const std::vector &training_input, const std::vector &training_output, const std::vector &test_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int m_tiles, int m_tile_size, int n_regressors, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Compute the predictions with full covariance matrix. @@ -84,13 +90,13 @@ std::vector> predict_with_full_cov( const std::vector &training_input, const std::vector &training_output, const std::vector &test_data, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int m_tiles, int m_tile_size, int n_regressors, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Compute loss for given data and Gaussian process model @@ -107,11 +113,11 @@ std::vector> predict_with_full_cov( */ double compute_loss(const std::vector &training_input, const std::vector &training_output, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int n_regressors, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Perform optimization for a given number of iterations @@ -137,10 +143,10 @@ optimize(const std::vector &training_input, int n_tiles, int n_tile_size, int n_regressors, - const gprat_hyper::AdamParams &adam_params, - gprat_hyper::SEKParams &sek_params, + const AdamParams &adam_params, + SEKParams &sek_params, std::vector trainable_params, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Perform a single optimization step @@ -166,11 +172,11 @@ double optimize_step(const std::vector &training_input, int n_tiles, int n_tile_size, int n_regressors, - gprat_hyper::AdamParams &adam_params, - gprat_hyper::SEKParams &sek_params, + AdamParams &adam_params, + SEKParams &sek_params, std::vector trainable_params, int iter, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Perform Cholesky decompositon (+ Assembly) @@ -188,12 +194,14 @@ double optimize_step(const std::vector &training_input, */ std::vector> cholesky(const std::vector &training_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int n_regressors, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); } // end of namespace gpu +GPRAT_NS_END + #endif diff --git a/core/include/gprat/gpu/gp_optimizer.cuh b/core/include/gprat/gpu/gp_optimizer.cuh index d0c5dd3a..ebe3aa43 100644 --- a/core/include/gprat/gpu/gp_optimizer.cuh +++ b/core/include/gprat/gpu/gp_optimizer.cuh @@ -1,12 +1,19 @@ -#ifndef GPU_GP_OPTIMIZER_H -#define GPU_GP_OPTIMIZER_H +#ifndef GPRAT_GPU_GP_OPTIMIZER_HPP +#define GPRAT_GPU_GP_OPTIMIZER_HPP + +#pragma once + +#include "gprat/detail/config.hpp" + +#include "gprat/gp_hyperparameters.hpp" +#include "gprat/gp_kernels.hpp" +#include "gprat/target.hpp" -#include "gp_hyperparameters.hpp" -#include "gp_kernels.hpp" -#include "target.hpp" #include #include +GPRAT_NS_BEGIN + namespace gpu { @@ -56,7 +63,7 @@ double compute_sigmoid(const double parameter); double compute_covariance_distance(std::size_t i_global, std::size_t j_global, std::size_t n_regressors, - gprat_hyper::SEKParams sek_params, + SEKParams sek_params, const std::vector &i_input, const std::vector &j_input); @@ -77,7 +84,7 @@ std::vector gen_tile_distance( std::size_t col, std::size_t N, std::size_t n_regressors, - gprat_hyper::SEKParams sek_params, + SEKParams sek_params, const std::vector &input); /** @@ -96,7 +103,7 @@ std::vector gen_tile_covariance_with_distance( std::size_t col, std::size_t N, std::size_t n_regressors, - gprat_hyper::SEKParams sek_params, + SEKParams sek_params, const std::vector &cov_dists); /** @@ -116,7 +123,7 @@ gen_tile_grad_v(std::size_t row, std::size_t col, std::size_t N, std::size_t n_regressors, - gprat_hyper::SEKParams sek_params, + SEKParams sek_params, const std::vector &cov_dists); /** @@ -136,7 +143,7 @@ gen_tile_grad_l(std::size_t row, std::size_t col, std::size_t N, std::size_t n_regressors, - gprat_hyper::SEKParams sek_params, + SEKParams sek_params, const std::vector &cov_dists); /** @@ -159,7 +166,7 @@ std::vector gen_tile_grad_v_trans(std::size_t N, const std::vector -gen_tile_grad_l_trans(std::size_t N, const hpx::shared_future f_grad_l_tile, gprat::CUDA_GPU &gpu); +gen_tile_grad_l_trans(std::size_t N, const hpx::shared_future f_grad_l_tile, CUDA_GPU &gpu); /** * @brief Compute hyper-parameter beta_1 or beta_2 to power t. @@ -187,7 +194,7 @@ compute_loss(const hpx::shared_future &K_diag_tile, const hpx::shared_future &alpha_tile, const hpx::shared_future &y_tile, std::size_t N, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Add up negative-log likelihood loss for all tiles. @@ -260,8 +267,8 @@ double update_second_moment(const double &gradient, double v_T, const double &be */ hpx::shared_future update_param(const double unconstrained_hyperparam, - gprat_hyper::SEKParams sek_params, - gprat_hyper::AdamParams adam_params, + SEKParams sek_params, + AdamParams adam_params, double m_T, double v_T, const std::vector beta1_T, @@ -319,7 +326,7 @@ sum_gradright(const std::vector &inter_alpha, const std::vector */ double sum_noise_gradleft(const std::vector &ft_invK, double grad, - gprat_hyper::SEKParams sek_params, + SEKParams sek_params, std::size_t N, std::size_t n_tiles); @@ -334,8 +341,10 @@ double sum_noise_gradleft(const std::vector &ft_invK, * @return The sum of the noise gradient */ double -sum_noise_gradright(const std::vector &alpha, double grad, gprat_hyper::SEKParams sek_params, std::size_t N); +sum_noise_gradright(const std::vector &alpha, double grad, SEKParams sek_params, std::size_t N); } // end of namespace gpu -#endif // end of GPU_GP_OPTIMIZER_H +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/gpu/gp_uncertainty.cuh b/core/include/gprat/gpu/gp_uncertainty.cuh index 8c2dce18..4a93eccb 100644 --- a/core/include/gprat/gpu/gp_uncertainty.cuh +++ b/core/include/gprat/gpu/gp_uncertainty.cuh @@ -1,7 +1,13 @@ -#ifndef GPU_GP_UNCERTAINTY_H -#define GPU_GP_UNCERTAINTY_H +#ifndef GPRAT_GPU_GP_UNCERTAINTY_HPP +#define GPRAT_GPU_GP_UNCERTAINTY_HPP -#include "target.hpp" +#pragma once + +#include "gprat/detail/config.hpp" + +#include "gprat/target.hpp" + +GPRAT_NS_BEGIN namespace gpu { @@ -16,7 +22,7 @@ namespace gpu * @return Diagonal elements of posterior covariance matrix */ hpx::shared_future diag_posterior( - const hpx::shared_future A, const hpx::shared_future B, std::size_t M, gprat::CUDA_GPU &gpu); + const hpx::shared_future A, const hpx::shared_future B, std::size_t M, CUDA_GPU &gpu); /** * @brief Retrieve diagonal elements of posterior covariance matrix. @@ -26,8 +32,10 @@ hpx::shared_future diag_posterior( * * @return Diagonal elements of posterior covariance matrix */ -hpx::shared_future diag_tile(const hpx::shared_future A, std::size_t M, gprat::CUDA_GPU &gpu); +hpx::shared_future diag_tile(const hpx::shared_future A, std::size_t M, CUDA_GPU &gpu); } // end of namespace gpu -#endif // end of GPU_GP_UNCERTAINTY_H +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/gpu/tiled_algorithms.cuh b/core/include/gprat/gpu/tiled_algorithms.cuh index 78c6f5cb..cc850679 100644 --- a/core/include/gprat/gpu/tiled_algorithms.cuh +++ b/core/include/gprat/gpu/tiled_algorithms.cuh @@ -1,12 +1,19 @@ -#ifndef GPU_TILED_ALGORITHMS_H -#define GPU_TILED_ALGORITHMS_H +#ifndef GPRAT_GPU_TILED_ALGORITHMS_HPP +#define GPRAT_GPU_TILED_ALGORITHMS_HPP + +#pragma once + +#include "gprat/detail/config.hpp" + +#include "gprat/gp_hyperparameters.hpp" +#include "gprat/target.hpp" +#include "gprat/gp_kernels.hpp" -#include "gp_hyperparameters.hpp" -#include "target.hpp" #include -#include #include +GPRAT_NS_BEGIN + namespace gpu { @@ -26,7 +33,7 @@ namespace gpu void right_looking_cholesky_tiled(std::vector> &ft_tiles, const std::size_t n_tile_size, const std::size_t n_tiles, - gprat::CUDA_GPU &gpu, + CUDA_GPU &gpu, const cusolverDnHandle_t &cusolver); // Tiled Triangular Solve Algorithms @@ -44,7 +51,7 @@ void forward_solve_tiled(std::vector> &ft_tiles, std::vector> &ft_rhs, const std::size_t n_tile_size, const std::size_t n_tiles, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Perform tiled backward triangular matrix-vector solve. @@ -59,7 +66,7 @@ void backward_solve_tiled(std::vector> &ft_tiles, std::vector> &ft_rhs, const std::size_t n_tile_size, const std::size_t n_tiles, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Perform tiled forward triangular matrix-matrix solve. @@ -79,7 +86,7 @@ void forward_solve_tiled_matrix( const std::size_t m_tile_size, const std::size_t n_tiles, const std::size_t m_tiles, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Perform tiled backward triangular matrix-matrix solve. @@ -99,7 +106,7 @@ void backward_solve_tiled_matrix( const std::size_t m_tile_size, const std::size_t n_tiles, const std::size_t m_tiles, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Perform tiled matrix-vector multiplication @@ -120,7 +127,7 @@ void matrix_vector_tiled(std::vector> &ft_tiles, const std::size_t N_col, const std::size_t n_tiles, const std::size_t m_tiles, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Perform tiled symmetric k-rank update on diagonal tiles @@ -140,14 +147,14 @@ void symmetric_matrix_matrix_diagonal_tiled( const std::size_t m_tile_size, const std::size_t n_tiles, const std::size_t m_tiles, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); void compute_gemm_of_invK_y(std::vector> &ft_invK, std::vector> &ft_y, std::vector> &ft_alpha, const std::size_t n_tile_size, const std::size_t n_tiles, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); // Tiled Loss hpx::shared_future compute_loss_tiled( @@ -156,7 +163,7 @@ hpx::shared_future compute_loss_tiled( std::vector> &ft_y, const std::size_t n_tile_size, const std::size_t n_tiles, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); // Tiled Diagonal of Posterior Covariance Matrix void symmetric_matrix_matrix_tiled( @@ -166,7 +173,7 @@ void symmetric_matrix_matrix_tiled( const std::size_t m_tile_size, const std::size_t n_tiles, const std::size_t m_tiles, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Compute the difference between two tiled vectors @@ -183,14 +190,14 @@ void vector_difference_tiled(std::vector> &ft_prior std::vector> &ft_vector, const std::size_t m_tile_size, const std::size_t m_tiles, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); // Tiled Prediction Uncertainty void matrix_diagonal_tiled(std::vector> &ft_priorK, std::vector> &ft_vector, const std::size_t m_tile_size, const std::size_t m_tiles, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); // Compute I-y*y^T*inv(K) void update_grad_K_tiled_mkl(std::vector> &ft_tiles, @@ -198,7 +205,7 @@ void update_grad_K_tiled_mkl(std::vector> &ft_tiles const std::vector> &ft_v2, const std::size_t n_tile_size, const std::size_t n_tiles, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Updates the lengthscale hyperparameter of the SEK kernel using Adam. @@ -223,8 +230,8 @@ double update_lengthscale( const std::vector> &ft_invK, const std::vector> &ft_gradparam, const std::vector> &ft_alpha, - gprat_hyper::SEKParams sek_params, - gprat_hyper::AdamParams adam_params, + SEKParams sek_params, + AdamParams adam_params, const std::size_t n_tile_size, const std::size_t n_tiles, std::vector> &m_T, @@ -232,7 +239,7 @@ double update_lengthscale( const std::vector> &beta1_T, const std::vector> &beta2_T, int iter, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Updates the vertical lengthscale hyperparameter of the SEK kernel @@ -258,8 +265,8 @@ double update_vertical_lengthscale( const std::vector> &ft_invK, const std::vector> &ft_gradparam, const std::vector> &ft_alpha, - gprat_hyper::SEKParams sek_params, - gprat_hyper::AdamParams adam_params, + SEKParams sek_params, + AdamParams adam_params, const std::size_t n_tile_size, const std::size_t n_tiles, std::vector> &m_T, @@ -267,7 +274,7 @@ double update_vertical_lengthscale( const std::vector> &beta1_T, const std::vector> &beta2_T, int iter, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); /** * @brief Updates a hyperparameter of the SEK kernel using Adam @@ -290,8 +297,8 @@ double update_vertical_lengthscale( double update_noise_variance( const std::vector> &ft_invK, const std::vector> &ft_alpha, - gprat_hyper::SEKParams sek_params, - gprat_hyper::AdamParams adam_params, + SEKParams sek_params, + AdamParams adam_params, const std::size_t n_tile_size, const std::size_t n_tiles, std::vector> &m_T, @@ -299,8 +306,10 @@ double update_noise_variance( const std::vector> &beta1_T, const std::vector> &beta2_T, int iter, - gprat::CUDA_GPU &gpu); + CUDA_GPU &gpu); } // end of namespace gpu -#endif // end of GPU_TILED_ALGORITHMS_H +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/target.hpp b/core/include/gprat/target.hpp index 8b66cb0b..13487114 100644 --- a/core/include/gprat/target.hpp +++ b/core/include/gprat/target.hpp @@ -1,5 +1,9 @@ -#ifndef TARGET_H -#define TARGET_H +#ifndef GPRAT_TARGET_H +#define GPRAT_TARGET_H + +#pragma once + +#include "gprat/detail/config.hpp" #include @@ -8,8 +12,7 @@ #include #endif -namespace gprat -{ +GPRAT_NS_BEGIN /** * @brief This class represents the target on which to perform the Gaussian @@ -203,6 +206,6 @@ void print_available_gpus(); */ int gpu_count(); -} // namespace gprat +GPRAT_NS_END -#endif // end of TARGET_H +#endif diff --git a/core/include/gprat/utils_c.hpp b/core/include/gprat/utils_c.hpp index 591bb7ee..21eb7a72 100644 --- a/core/include/gprat/utils_c.hpp +++ b/core/include/gprat/utils_c.hpp @@ -1,5 +1,9 @@ -#ifndef UTILS_C_H -#define UTILS_C_H +#ifndef GPRAT_UTILS_C_H +#define GPRAT_UTILS_C_H + +#pragma once + +#include "gprat/detail/config.hpp" #include #include @@ -7,8 +11,8 @@ #include #include -namespace utils -{ +GPRAT_NS_BEGIN + /** * @brief Compute the number of tiles for training data, given the number of * samples and the size of each tile. @@ -85,6 +89,6 @@ void stop_hpx_runtime(); */ bool compiled_with_cuda(); -} // namespace utils +GPRAT_NS_END #endif diff --git a/core/src/cpu/adapter_cblas_fp32.cpp b/core/src/cpu/adapter_cblas_fp32.cpp index d91a3867..2b7e5c12 100644 --- a/core/src/cpu/adapter_cblas_fp32.cpp +++ b/core/src/cpu/adapter_cblas_fp32.cpp @@ -1,4 +1,4 @@ -#include "cpu/adapter_cblas_fp32.hpp" +#include "gprat/cpu/adapter_cblas_fp32.hpp" #ifdef GPRAT_ENABLE_MKL // MKL CBLAS and LAPACKE @@ -9,6 +9,8 @@ #include "lapacke.h" #endif +GPRAT_NS_BEGIN + // BLAS level 3 operations vector_future potrf(vector_future f_A, const int N) @@ -193,3 +195,5 @@ float dot(std::vector a, std::vector b, const int N) // DOT: a * b return cblas_sdot(N, a.data(), 1, b.data(), 1); } + +GPRAT_NS_END diff --git a/core/src/cpu/adapter_cblas_fp64.cpp b/core/src/cpu/adapter_cblas_fp64.cpp index 0c38b3c2..3cc15500 100644 --- a/core/src/cpu/adapter_cblas_fp64.cpp +++ b/core/src/cpu/adapter_cblas_fp64.cpp @@ -1,4 +1,4 @@ -#include "cpu/adapter_cblas_fp64.hpp" +#include "gprat/cpu/adapter_cblas_fp64.hpp" #ifdef GPRAT_ENABLE_MKL // MKL CBLAS and LAPACKE @@ -9,6 +9,8 @@ #include "lapacke.h" #endif +GPRAT_NS_BEGIN + // BLAS level 3 operations vector_future potrf(vector_future f_A, const int N) @@ -193,3 +195,5 @@ double dot(std::vector a, std::vector b, const int N) // DOT: a * b return cblas_ddot(N, a.data(), 1, b.data(), 1); } + +GPRAT_NS_END diff --git a/core/src/cpu/gp_algorithms.cpp b/core/src/cpu/gp_algorithms.cpp index 92193b6d..8b42e12a 100644 --- a/core/src/cpu/gp_algorithms.cpp +++ b/core/src/cpu/gp_algorithms.cpp @@ -1,8 +1,10 @@ -#include "cpu/gp_algorithms.hpp" +#include "gprat/cpu/gp_algorithms.hpp" #include #include +GPRAT_NS_BEGIN + namespace cpu { @@ -11,7 +13,7 @@ namespace cpu double compute_covariance_function(std::size_t i_global, std::size_t j_global, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, const std::vector &i_input, const std::vector &j_input) { @@ -32,7 +34,7 @@ std::vector gen_tile_covariance( std::size_t col, std::size_t N, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, const std::vector &input) { std::size_t i_global, j_global; @@ -66,7 +68,7 @@ std::vector gen_tile_full_prior_covariance( std::size_t col, std::size_t N, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, const std::vector &input) { std::size_t i_global, j_global; @@ -92,7 +94,7 @@ std::vector gen_tile_prior_covariance( std::size_t col, std::size_t N, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, const std::vector &input) { std::size_t i_global, j_global; @@ -116,7 +118,7 @@ std::vector gen_tile_cross_covariance( std::size_t N_row, std::size_t N_col, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, const std::vector &row_input, const std::vector &col_input) { @@ -204,3 +206,5 @@ double compute_error_norm(std::size_t n_tiles, } } // end of namespace cpu + +GPRAT_NS_END diff --git a/core/src/cpu/gp_functions.cpp b/core/src/cpu/gp_functions.cpp index 92caa275..9f16f319 100644 --- a/core/src/cpu/gp_functions.cpp +++ b/core/src/cpu/gp_functions.cpp @@ -1,10 +1,13 @@ -#include "cpu/gp_functions.hpp" +#include "gprat/cpu/gp_functions.hpp" + +#include "gprat/cpu/gp_algorithms.hpp" +#include "gprat/cpu/gp_optimizer.hpp" +#include "gprat/cpu/tiled_algorithms.hpp" -#include "cpu/gp_algorithms.hpp" -#include "cpu/gp_optimizer.hpp" -#include "cpu/tiled_algorithms.hpp" #include +GPRAT_NS_BEGIN + using Tiled_matrix = std::vector>>; using Tiled_vector = std::vector>>; @@ -15,7 +18,7 @@ namespace cpu // PREDICT std::vector> cholesky(const std::vector &training_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int n_regressors) @@ -66,7 +69,7 @@ std::vector predict(const std::vector &training_input, const std::vector &training_output, const std::vector &test_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int m_tiles, @@ -183,7 +186,7 @@ std::vector> predict_with_uncertainty( const std::vector &training_input, const std::vector &training_output, const std::vector &test_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int m_tiles, @@ -380,7 +383,7 @@ std::vector> predict_with_full_cov( const std::vector &training_input, const std::vector &training_output, const std::vector &test_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int m_tiles, @@ -589,7 +592,7 @@ std::vector> predict_with_full_cov( // OPTIMIZATION double compute_loss(const std::vector &training_input, const std::vector &training_output, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int n_regressors) @@ -676,8 +679,8 @@ optimize(const std::vector &training_input, int n_tiles, int n_tile_size, int n_regressors, - const gprat_hyper::AdamParams &adam_params, - gprat_hyper::SEKParams &sek_params, + const AdamParams &adam_params, + SEKParams &sek_params, std::vector trainable_params) { /* @@ -924,8 +927,8 @@ double optimize_step(const std::vector &training_input, int n_tiles, int n_tile_size, int n_regressors, - gprat_hyper::AdamParams &adam_params, - gprat_hyper::SEKParams &sek_params, + AdamParams &adam_params, + SEKParams &sek_params, std::vector trainable_params, int iter) { @@ -1159,3 +1162,5 @@ double optimize_step(const std::vector &training_input, } } // end of namespace cpu + +GPRAT_NS_END diff --git a/core/src/cpu/gp_optimizer.cpp b/core/src/cpu/gp_optimizer.cpp index d33b1889..081c037e 100644 --- a/core/src/cpu/gp_optimizer.cpp +++ b/core/src/cpu/gp_optimizer.cpp @@ -1,9 +1,12 @@ -#include "cpu/gp_optimizer.hpp" +#include "gprat/cpu/gp_optimizer.hpp" + +#include "gprat/cpu/adapter_cblas_fp64.hpp" -#include "cpu/adapter_cblas_fp64.hpp" #include #include +GPRAT_NS_BEGIN + namespace cpu { @@ -40,7 +43,7 @@ double compute_sigmoid(double parameter) { return 1.0 / (1.0 + exp(-parameter)); double compute_covariance_distance(std::size_t i_global, std::size_t j_global, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, const std::vector &i_input, const std::vector &j_input) { @@ -61,7 +64,7 @@ std::vector gen_tile_distance( std::size_t col, std::size_t N, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, const std::vector &input) { std::size_t i_global, j_global; @@ -82,11 +85,7 @@ std::vector gen_tile_distance( } std::vector gen_tile_covariance_with_distance( - std::size_t row, - std::size_t col, - std::size_t N, - const gprat_hyper::SEKParams &sek_params, - const std::vector &distance) + std::size_t row, std::size_t col, std::size_t N, const SEKParams &sek_params, const std::vector &distance) { std::size_t i_global, j_global; double covariance; @@ -112,8 +111,7 @@ std::vector gen_tile_covariance_with_distance( return tile; } -std::vector -gen_tile_grad_v(std::size_t N, const gprat_hyper::SEKParams &sek_params, const std::vector &distance) +std::vector gen_tile_grad_v(std::size_t N, const SEKParams &sek_params, const std::vector &distance) { // Preallocate required memory std::vector tile; @@ -130,8 +128,7 @@ gen_tile_grad_v(std::size_t N, const gprat_hyper::SEKParams &sek_params, const s return tile; } -std::vector -gen_tile_grad_l(std::size_t N, const gprat_hyper::SEKParams &sek_params, const std::vector &distance) +std::vector gen_tile_grad_l(std::size_t N, const SEKParams &sek_params, const std::vector &distance) { // Preallocate required memory std::vector tile; @@ -161,11 +158,8 @@ double update_second_moment(double gradient, double v_T, double beta_2) return beta_2 * v_T + (1.0 - beta_2) * gradient * gradient; } -double adam_step(const double unconstrained_hyperparam, - const gprat_hyper::AdamParams &adam_params, - double m_T, - double v_T, - std::size_t iter) +double adam_step( + const double unconstrained_hyperparam, const AdamParams &adam_params, double m_T, double v_T, std::size_t iter) { // Compute decay rate double beta1_T = pow(adam_params.beta1, static_cast(iter + 1)); @@ -245,3 +239,5 @@ double compute_trace_diag(const std::vector &tile, double trace, std::si } } // end of namespace cpu + +GPRAT_NS_END diff --git a/core/src/cpu/gp_uncertainty.cpp b/core/src/cpu/gp_uncertainty.cpp index 3ea6a7a9..a0cf4511 100644 --- a/core/src/cpu/gp_uncertainty.cpp +++ b/core/src/cpu/gp_uncertainty.cpp @@ -1,4 +1,6 @@ -#include "cpu/gp_uncertainty.hpp" +#include "gprat/cpu/gp_uncertainty.hpp" + +GPRAT_NS_BEGIN namespace cpu { @@ -19,3 +21,5 @@ hpx::shared_future> get_matrix_diagonal(hpx::shared_future +GPRAT_NS_BEGIN + namespace cpu { @@ -297,8 +300,8 @@ void update_hyperparameter_tiled( const Tiled_matrix &ft_invK, const Tiled_matrix &ft_gradK_param, const Tiled_vector &ft_alpha, - const gprat_hyper::AdamParams &adam_params, - gprat_hyper::SEKParams &sek_params, + const AdamParams &adam_params, + SEKParams &sek_params, int N, std::size_t n_tiles, std::size_t iter, @@ -448,3 +451,5 @@ void update_hyperparameter_tiled( } } // end of namespace cpu + +GPRAT_NS_END diff --git a/core/src/gp_hyperparameters.cpp b/core/src/gp_hyperparameters.cpp index f0a8caab..a44bdc36 100644 --- a/core/src/gp_hyperparameters.cpp +++ b/core/src/gp_hyperparameters.cpp @@ -1,10 +1,9 @@ -#include "gp_hyperparameters.hpp" +#include "gprat/gp_hyperparameters.hpp" #include #include -namespace gprat_hyper -{ +GPRAT_NS_BEGIN AdamParams::AdamParams(double lr, double b1, double b2, double eps, int opt_i) : learning_rate(lr), @@ -30,4 +29,4 @@ std::string AdamParams::repr() const return oss.str(); } -} // namespace gprat_hyper +GPRAT_NS_END diff --git a/core/src/gp_kernels.cpp b/core/src/gp_kernels.cpp index 42952e7e..66b789b4 100644 --- a/core/src/gp_kernels.cpp +++ b/core/src/gp_kernels.cpp @@ -1,13 +1,13 @@ -#include "gp_kernels.hpp" +#include "gprat/gp_kernels.hpp" #include -namespace gprat_hyper -{ -SEKParams::SEKParams(double lengthscale_, double vertical_lengthscale_, double noise_variance_) : - lengthscale(lengthscale_), - vertical_lengthscale(vertical_lengthscale_), - noise_variance(noise_variance_) +GPRAT_NS_BEGIN + +SEKParams::SEKParams(double lengthscale, double vertical_lengthscale, double noise_variance) : + lengthscale(lengthscale), + vertical_lengthscale(vertical_lengthscale), + noise_variance(noise_variance) { m_T.resize(this->size()); w_T.resize(this->size()); @@ -51,4 +51,5 @@ const double &SEKParams::get_param(std::size_t index) const } throw std::invalid_argument("Get Invalid param_idx"); } -} // namespace gprat_hyper + +GPRAT_NS_END diff --git a/core/src/gprat_c.cpp b/core/src/gprat_c.cpp index c93e792c..d7805804 100644 --- a/core/src/gprat_c.cpp +++ b/core/src/gprat_c.cpp @@ -1,23 +1,22 @@ -#include "gprat_c.hpp" +#include "gprat/gprat_c.hpp" -#include "cpu/gp_functions.hpp" -#include "utils_c.hpp" -#include +#include "gprat/cpu/gp_functions.hpp" +#include "gprat/utils_c.hpp" #if GPRAT_WITH_CUDA #include "gpu/gp_functions.cuh" #endif -// namespace for GPRat library entities -namespace gprat -{ +#include + +GPRAT_NS_BEGIN GP_data::GP_data(const std::string &f_path, int n, int n_reg) : file_path(f_path), n_samples(n), n_regressors(n_reg) { - data = utils::load_data(f_path, n, n_reg - 1); + data = load_data(f_path, n, n_reg - 1); } GP::GP(std::vector input, @@ -121,7 +120,7 @@ std::vector GP::predict(const std::vector &test_input, int m_til m_tiles, m_tile_size, n_reg, - *std::dynamic_pointer_cast(target_)); + *std::dynamic_pointer_cast(target_)); } else { @@ -171,7 +170,7 @@ GP::predict_with_uncertainty(const std::vector &test_input, int m_tiles, m_tiles, m_tile_size, n_reg, - *std::dynamic_pointer_cast(target_)); + *std::dynamic_pointer_cast(target_)); } else { @@ -221,7 +220,7 @@ GP::predict_with_full_cov(const std::vector &test_input, int m_tiles, in m_tiles, m_tile_size, n_reg, - *std::dynamic_pointer_cast(target_)); + *std::dynamic_pointer_cast(target_)); } else { @@ -252,7 +251,7 @@ GP::predict_with_full_cov(const std::vector &test_input, int m_tiles, in .get(); } -std::vector GP::optimize(const gprat_hyper::AdamParams &adam_params) +std::vector GP::optimize(const AdamParams &adam_params) { return hpx::async( [this, &adam_params]() @@ -277,7 +276,7 @@ std::vector GP::optimize(const gprat_hyper::AdamParams &adam_params) .get(); } -double GP::optimize_step(gprat_hyper::AdamParams &adam_params, int iter) +double GP::optimize_step(AdamParams &adam_params, int iter) { return hpx::async( [this, &adam_params, iter]() @@ -318,7 +317,7 @@ double GP::calculate_loss() n_tiles_, n_tile_size_, n_reg, - *std::dynamic_pointer_cast(target_)); + *std::dynamic_pointer_cast(target_)); } else { @@ -347,7 +346,7 @@ std::vector> GP::cholesky() n_tiles_, n_tile_size_, n_reg, - *std::dynamic_pointer_cast(target_)); + *std::dynamic_pointer_cast(target_)); } else { @@ -360,4 +359,4 @@ std::vector> GP::cholesky() .get(); } -} // namespace gprat +GPRAT_NS_END diff --git a/core/src/gpu/adapter_cublas.cu b/core/src/gpu/adapter_cublas.cu index 61227e8d..c3833aac 100644 --- a/core/src/gpu/adapter_cublas.cu +++ b/core/src/gpu/adapter_cublas.cu @@ -1,4 +1,6 @@ -#include "gpu/adapter_cublas.cuh" +#include "gprat/gpu/adapter_cublas.cuh" + +GPRAT_NS_BEGIN // frequently used names using hpx::cuda::experimental::check_cuda_error; @@ -411,3 +413,5 @@ dot(cublasHandle_t cublas, return hpx::make_ready_future(result); } + +GPRAT_NS_END diff --git a/core/src/gpu/cuda_kernels.cu b/core/src/gpu/cuda_kernels.cu index 37378f37..5e77ec6a 100644 --- a/core/src/gpu/cuda_kernels.cu +++ b/core/src/gpu/cuda_kernels.cu @@ -1,6 +1,8 @@ -#include "gpu/cuda_kernels.cuh" +#include "gprat/gpu/cuda_kernels.cuh" -#include "gpu/cuda_utils.cuh" +#include "gprat/gpu/cuda_utils.cuh" + +GPRAT_NS_BEGIN __global__ void transpose(double *transposed, double *original, std::size_t width, std::size_t height) { @@ -25,3 +27,5 @@ __global__ void transpose(double *transposed, double *original, std::size_t widt transposed[index_out] = block[threadIdx.x][threadIdx.y]; } } + +GPRAT_NS_END diff --git a/core/src/gpu/gp_algorithms.cu b/core/src/gpu/gp_algorithms.cu index 39407ed6..832450fe 100644 --- a/core/src/gpu/gp_algorithms.cu +++ b/core/src/gpu/gp_algorithms.cu @@ -1,14 +1,17 @@ -#include "gpu/gp_algorithms.cuh" +#include "gprat/gpu/gp_algorithms.cuh" + +#include "gprat/gp_kernels.hpp" +#include "gprat/gpu/cuda_kernels.cuh" +#include "gprat/gpu/cuda_utils.cuh" +#include "gprat/gpu/gp_optimizer.cuh" +#include "gprat/target.hpp" -#include "gp_kernels.hpp" -#include "gpu/cuda_kernels.cuh" -#include "gpu/cuda_utils.cuh" -#include "gpu/gp_optimizer.cuh" -#include "target.hpp" #include #include #include +GPRAT_NS_BEGIN + namespace gpu { @@ -20,7 +23,7 @@ __global__ void gen_tile_covariance_kernel( const std::size_t n_regressors, const std::size_t tile_row, const std::size_t tile_column, - const gprat_hyper::SEKParams sek_params) + const SEKParams sek_params) { // Compute the global indices of the thread unsigned int i = blockIdx.y * blockDim.y + threadIdx.y; @@ -59,8 +62,8 @@ double *gen_tile_covariance(const double *d_input, const std::size_t tile_column, const std::size_t n_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu) + const SEKParams sek_params, + CUDA_GPU &gpu) { double *d_tile; @@ -85,7 +88,7 @@ __global__ void gen_tile_full_prior_covariance_kernel( const std::size_t n_regressors, const std::size_t tile_row, const std::size_t tile_column, - const gprat_hyper::SEKParams sek_params) + const SEKParams sek_params) { unsigned int i = blockIdx.y * blockDim.y + threadIdx.y; unsigned int j = blockIdx.x * blockDim.x + threadIdx.x; @@ -117,8 +120,8 @@ double *gen_tile_full_prior_covariance( const std::size_t tile_colums, const std::size_t n_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu) + const SEKParams sek_params, + CUDA_GPU &gpu) { double *d_tile; @@ -143,7 +146,7 @@ __global__ void gen_tile_prior_covariance_kernel( const std::size_t n_regressors, const std::size_t tile_row, const std::size_t tile_column, - const gprat_hyper::SEKParams sek_params) + const SEKParams sek_params) { unsigned int i = blockIdx.x * blockDim.x + threadIdx.x; @@ -174,8 +177,8 @@ double *gen_tile_prior_covariance( const std::size_t tile_column, const std::size_t n_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu) + const SEKParams sek_params, + CUDA_GPU &gpu) { double *d_tile; @@ -202,7 +205,7 @@ __global__ void gen_tile_cross_covariance_kernel( const std::size_t tile_row, const std::size_t tile_column, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params) + const SEKParams sek_params) { unsigned int i = blockIdx.y * blockDim.y + threadIdx.y; unsigned int j = blockIdx.x * blockDim.x + threadIdx.x; @@ -235,8 +238,8 @@ double *gen_tile_cross_covariance( const std::size_t n_row_tile_size, const std::size_t n_column_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu) + const SEKParams sek_params, + CUDA_GPU &gpu) { double *d_tile; @@ -265,7 +268,7 @@ double *gen_tile_cross_covariance( hpx::shared_future gen_tile_cross_cov_T(std::size_t n_row_tile_size, std::size_t n_column_tile_size, const hpx::shared_future f_cross_covariance_tile, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { double *transposed; check_cuda_error(cudaMalloc(&transposed, n_row_tile_size * n_column_tile_size * sizeof(double))); @@ -293,8 +296,7 @@ __global__ void gen_tile_output_kernel(double *tile, const double *output, std:: } } -double * -gen_tile_output(const std::size_t row, const std::size_t n_tile_size, const double *d_output, gprat::CUDA_GPU &gpu) +double *gen_tile_output(const std::size_t row, const std::size_t n_tile_size, const double *d_output, CUDA_GPU &gpu) { dim3 threads_per_block(256); dim3 n_blocks((n_tile_size + 255) / 256); @@ -311,7 +313,7 @@ gen_tile_output(const std::size_t row, const std::size_t n_tile_size, const doub return d_tile; } -double *gen_tile_zeros(std::size_t n_tile_size, gprat::CUDA_GPU &gpu) +double *gen_tile_zeros(std::size_t n_tile_size, CUDA_GPU &gpu) { double *d_tile; cudaStream_t stream = gpu.next_stream(); @@ -345,8 +347,8 @@ std::vector> assemble_tiled_covariance_matrix( const std::size_t n_tiles, const std::size_t n_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu) + const SEKParams sek_params, + CUDA_GPU &gpu) { std::vector> d_tiles(n_tiles * n_tiles); @@ -369,8 +371,8 @@ std::vector> assemble_tiled_covariance_matrix( return d_tiles; } -std::vector> assemble_alpha_tiles( - const double *d_output, const std::size_t n_tiles, const std::size_t n_tile_size, gprat::CUDA_GPU &gpu) +std::vector> +assemble_alpha_tiles(const double *d_output, const std::size_t n_tiles, const std::size_t n_tile_size, CUDA_GPU &gpu) { std::vector> alpha_tiles(n_tiles); for (std::size_t i = 0; i < n_tiles; i++) @@ -390,8 +392,8 @@ std::vector> assemble_cross_covariance_tiles( const std::size_t m_tile_size, const std::size_t n_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu) + const SEKParams sek_params, + CUDA_GPU &gpu) { std::vector> cross_covariance_tiles; cross_covariance_tiles.resize(m_tiles * n_tiles); @@ -416,7 +418,7 @@ std::vector> assemble_cross_covariance_tiles( } std::vector> -assemble_tiles_with_zeros(std::size_t n_tile_size, std::size_t n_tiles, gprat::CUDA_GPU &gpu) +assemble_tiles_with_zeros(std::size_t n_tile_size, std::size_t n_tiles, CUDA_GPU &gpu) { std::vector> tiles(n_tiles); for (std::size_t i = 0; i < n_tiles; i++) @@ -431,8 +433,8 @@ std::vector> assemble_prior_K_tiles( const std::size_t m_tiles, const std::size_t m_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu) + const SEKParams sek_params, + CUDA_GPU &gpu) { std::vector> d_prior_K_tiles; d_prior_K_tiles.resize(m_tiles); @@ -449,8 +451,8 @@ std::vector> assemble_prior_K_tiles_full( const std::size_t m_tiles, const std::size_t m_tile_size, const std::size_t n_regressors, - const gprat_hyper::SEKParams sek_params, - gprat::CUDA_GPU &gpu) + const SEKParams sek_params, + CUDA_GPU &gpu) { std::vector> d_prior_K_tiles(m_tiles * m_tiles); for (std::size_t i = 0; i < m_tiles; i++) @@ -483,7 +485,7 @@ std::vector> assemble_t_cross_covariance_tiles( const std::size_t m_tiles, const std::size_t n_tile_size, const std::size_t m_tile_size, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { std::vector> d_t_cross_covariance_tiles(m_tiles * n_tiles); for (std::size_t i = 0; i < m_tiles; i++) @@ -502,7 +504,7 @@ std::vector> assemble_t_cross_covariance_tiles( } std::vector> assemble_y_tiles( - const double *d_training_output, const std::size_t n_tiles, const std::size_t n_tile_size, gprat::CUDA_GPU &gpu) + const double *d_training_output, const std::size_t n_tiles, const std::size_t n_tile_size, CUDA_GPU &gpu) { std::vector> d_y_tiles(n_tiles); for (std::size_t i = 0; i < n_tiles; i++) @@ -512,10 +514,8 @@ std::vector> assemble_y_tiles( return d_y_tiles; } -std::vector copy_tiled_vector_to_host_vector(std::vector> &d_tiles, - std::size_t n_tile_size, - std::size_t n_tiles, - gprat::CUDA_GPU &gpu) +std::vector copy_tiled_vector_to_host_vector( + std::vector> &d_tiles, std::size_t n_tile_size, std::size_t n_tiles, CUDA_GPU &gpu) { std::vector h_vector(n_tiles * n_tile_size); std::vector streams(n_tiles); @@ -537,7 +537,7 @@ std::vector> move_lower_tiled_matrix_to_host( const std::vector> &d_tiles, const std::size_t n_tile_size, const std::size_t n_tiles, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { std::vector> h_tiles(n_tiles * n_tiles); @@ -574,3 +574,5 @@ void free_lower_tiled_matrix(const std::vector> &d_ } } // end of namespace gpu + +GPRAT_NS_END diff --git a/core/src/gpu/gp_functions.cu b/core/src/gpu/gp_functions.cu index 8f5e341f..7ba8142b 100644 --- a/core/src/gpu/gp_functions.cu +++ b/core/src/gpu/gp_functions.cu @@ -1,14 +1,17 @@ -#include "gpu/gp_functions.cuh" +#include "gprat/gpu/gp_functions.cuh" + +#include "gprat/gp_kernels.hpp" +#include "gprat/gpu/cuda_utils.cuh" +#include "gprat/gpu/gp_algorithms.cuh" +#include "gprat/gpu/tiled_algorithms.cuh" +#include "gprat/target.hpp" -#include "gp_kernels.hpp" -#include "gpu/cuda_utils.cuh" -#include "gpu/gp_algorithms.cuh" -#include "gpu/tiled_algorithms.cuh" -#include "target.hpp" #include #include #include +GPRAT_NS_BEGIN + namespace gpu { @@ -16,13 +19,13 @@ std::vector predict(const std::vector &h_training_input, const std::vector &h_training_output, const std::vector &h_test_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int m_tiles, int m_tile_size, int n_regressors, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { gpu.create(); @@ -65,13 +68,13 @@ std::vector> predict_with_uncertainty( const std::vector &h_training_input, const std::vector &h_training_output, const std::vector &h_test_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int m_tiles, int m_tile_size, int n_regressors, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { gpu.create(); @@ -150,13 +153,13 @@ std::vector> predict_with_full_cov( const std::vector &h_training_input, const std::vector &h_training_output, const std::vector &h_test_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int m_tiles, int m_tile_size, int n_regressors, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { gpu.create(); @@ -229,11 +232,11 @@ std::vector> predict_with_full_cov( double compute_loss(const std::vector &h_training_input, const std::vector &h_training_output, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int n_regressors, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { gpu.create(); @@ -279,10 +282,10 @@ optimize(const std::vector &training_input, int n_tiles, int n_tile_size, int n_regressors, - const gprat_hyper::AdamParams &adam_params, - const gprat_hyper::SEKParams &sek_params, + const AdamParams &adam_params, + const SEKParams &sek_params, std::vector trainable_params, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { throw std::logic_error("Function not implemented for GPU"); // return std::vector>(); @@ -293,11 +296,11 @@ double optimize_step(const std::vector &training_input, int n_tiles, int n_tile_size, int n_regressors, - gprat_hyper::AdamParams &adam_params, - gprat_hyper::SEKParams &sek_params, + AdamParams &adam_params, + SEKParams &sek_params, std::vector trainable_params, int iter, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { throw std::logic_error("Function not implemented for GPU"); // return 0.0; @@ -305,11 +308,11 @@ double optimize_step(const std::vector &training_input, std::vector> cholesky(const std::vector &h_training_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, int n_tiles, int n_tile_size, int n_regressors, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { gpu.create(); @@ -333,3 +336,5 @@ cholesky(const std::vector &h_training_input, } } // end of namespace gpu + +GPRAT_NS_END diff --git a/core/src/gpu/gp_optimizer.cu b/core/src/gpu/gp_optimizer.cu index 53cca8bb..62727414 100644 --- a/core/src/gpu/gp_optimizer.cu +++ b/core/src/gpu/gp_optimizer.cu @@ -1,8 +1,10 @@ -#include "gpu/gp_optimizer.cuh" +#include "gprat/gpu/gp_optimizer.cuh" -#include "gpu/adapter_cublas.cuh" -#include "gpu/cuda_kernels.cuh" -#include "gpu/cuda_utils.cuh" +#include "gprat/gpu/adapter_cublas.cuh" +#include "gprat/gpu/cuda_kernels.cuh" +#include "gprat/gpu/cuda_utils.cuh" + +GPRAT_NS_BEGIN namespace gpu { @@ -36,7 +38,7 @@ double compute_sigmoid(const double parameter) { return 1.0 / (1.0 + exp(-parame double compute_covariance_distance(std::size_t i_global, std::size_t j_global, std::size_t n_regressors, - gprat_hyper::SEKParams sek_params, + SEKParams sek_params, const std::vector &i_input, const std::vector &j_input) { @@ -58,7 +60,7 @@ std::vector gen_tile_distance( std::size_t col, std::size_t N, std::size_t n_regressors, - gprat_hyper::SEKParams sek_params, + SEKParams sek_params, const std::vector &input) { std::size_t i_global, j_global; @@ -85,7 +87,7 @@ std::vector gen_tile_covariance_with_distance( std::size_t col, std::size_t N, std::size_t n_regressors, - gprat_hyper::SEKParams sek_params, + SEKParams sek_params, const std::vector &cov_dists) { std::size_t i_global, j_global; @@ -117,7 +119,7 @@ gen_tile_grad_v(std::size_t row, std::size_t col, std::size_t N, std::size_t n_regressors, - gprat_hyper::SEKParams sek_params, + SEKParams sek_params, const std::vector &cov_dists) { // Initialize tile @@ -140,7 +142,7 @@ gen_tile_grad_l(std::size_t row, std::size_t col, std::size_t N, std::size_t n_regressors, - gprat_hyper::SEKParams sek_params, + SEKParams sek_params, const std::vector &cov_dists) { // Initialize tile @@ -176,7 +178,7 @@ std::vector gen_tile_grad_v_trans(std::size_t N, const std::vector -gen_tile_grad_l_trans(std::size_t N, const hpx::shared_future f_grad_l_tile, gprat::CUDA_GPU &gpu) +gen_tile_grad_l_trans(std::size_t N, const hpx::shared_future f_grad_l_tile, CUDA_GPU &gpu) { double *transposed; check_cuda_error(cudaMalloc(&transposed, N * N * sizeof(double))); @@ -209,7 +211,7 @@ compute_loss(const hpx::shared_future &K_diag_tile, const hpx::shared_future &alpha_tile, const hpx::shared_future &y_tile, std::size_t N, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { auto [cublas, stream] = gpu.next_cublas_handle(); @@ -276,8 +278,8 @@ double update_second_moment(const double &gradient, double v_T, const double &be hpx::shared_future update_param(const double unconstrained_hyperparam, - gprat_hyper::SEKParams sek_params, - gprat_hyper::AdamParams adam_params, + SEKParams sek_params, + AdamParams adam_params, double m_T, double v_T, const std::vector beta1_T, @@ -339,11 +341,8 @@ sum_gradright(const std::vector &inter_alpha, const std::vector return 0.0; } -double sum_noise_gradleft(const std::vector &ft_invK, - double grad, - gprat_hyper::SEKParams sek_params, - std::size_t N, - std::size_t n_tiles) +double sum_noise_gradleft( + const std::vector &ft_invK, double grad, SEKParams sek_params, std::size_t N, std::size_t n_tiles) { double noise_der = compute_sigmoid(to_unconstrained(sek_params.noise_variance, true)); for (std::size_t i = 0; i < N; ++i) @@ -353,8 +352,7 @@ double sum_noise_gradleft(const std::vector &ft_invK, return std::move(grad); } -double -sum_noise_gradright(const std::vector &alpha, double grad, gprat_hyper::SEKParams sek_params, std::size_t N) +double sum_noise_gradright(const std::vector &alpha, double grad, SEKParams sek_params, std::size_t N) { // double noise_der = // compute_sigmoid(to_unconstrained(sek_params.noise_variance, true)); @@ -364,3 +362,5 @@ sum_noise_gradright(const std::vector &alpha, double grad, gprat_hyper:: } } // end of namespace gpu + +GPRAT_NS_END diff --git a/core/src/gpu/gp_uncertainty.cu b/core/src/gpu/gp_uncertainty.cu index a7919457..6cc7f50b 100644 --- a/core/src/gpu/gp_uncertainty.cu +++ b/core/src/gpu/gp_uncertainty.cu @@ -1,16 +1,19 @@ -#include "gpu/gp_uncertainty.cuh" +#include "gprat/gpu/gp_uncertainty.cuh" + +#include "gprat/gpu/cuda_utils.cuh" +#include "gprat/target.hpp" -#include "gpu/cuda_utils.cuh" -#include "target.hpp" #include +GPRAT_NS_BEGIN + using hpx::cuda::experimental::check_cuda_error; namespace gpu { -hpx::shared_future diag_posterior( - const hpx::shared_future A, const hpx::shared_future B, std::size_t M, gprat::CUDA_GPU &gpu) +hpx::shared_future +diag_posterior(const hpx::shared_future A, const hpx::shared_future B, std::size_t M, CUDA_GPU &gpu) { auto [cublas, stream] = gpu.next_cublas_handle(); @@ -27,7 +30,7 @@ hpx::shared_future diag_posterior( return hpx::make_ready_future(tile); } -hpx::shared_future diag_tile(const hpx::shared_future A, std::size_t M, gprat::CUDA_GPU &gpu) +hpx::shared_future diag_tile(const hpx::shared_future A, std::size_t M, CUDA_GPU &gpu) { double *diag_tile; check_cuda_error(cudaMalloc(&diag_tile, M * sizeof(double))); @@ -41,3 +44,5 @@ hpx::shared_future diag_tile(const hpx::shared_future A, std } } // end of namespace gpu + +GPRAT_NS_END diff --git a/core/src/gpu/tiled_algorithms.cu b/core/src/gpu/tiled_algorithms.cu index 1ffdd866..3c479ffd 100644 --- a/core/src/gpu/tiled_algorithms.cu +++ b/core/src/gpu/tiled_algorithms.cu @@ -1,10 +1,13 @@ -#include "gpu/tiled_algorithms.cuh" +#include "gprat/gpu/tiled_algorithms.cuh" + +#include "gprat/gpu/adapter_cublas.cuh" +#include "gprat/gpu/gp_optimizer.cuh" +#include "gprat/gpu/gp_uncertainty.cuh" -#include "gpu/adapter_cublas.cuh" -#include "gpu/gp_optimizer.cuh" -#include "gpu/gp_uncertainty.cuh" #include +GPRAT_NS_BEGIN + namespace gpu { @@ -13,7 +16,7 @@ namespace gpu void right_looking_cholesky_tiled(std::vector> &ft_tiles, const std::size_t n_tile_size, const std::size_t n_tiles, - gprat::CUDA_GPU &gpu, + CUDA_GPU &gpu, const cusolverDnHandle_t &cusolver) { for (std::size_t k = 0; k < n_tiles; ++k) @@ -86,7 +89,7 @@ void forward_solve_tiled(std::vector> &ft_tiles, std::vector> &ft_rhs, const std::size_t n_tile_size, const std::size_t n_tiles, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { for (std::size_t k = 0; k < n_tiles; ++k) { @@ -120,7 +123,7 @@ void backward_solve_tiled(std::vector> &ft_tiles, std::vector> &ft_rhs, const std::size_t n_tile_size, const std::size_t n_tiles, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { // NOTE: The loops traverse backwards. Its last comparisons require the // usage negative numbers. Therefore they use signed int instead of the @@ -160,7 +163,7 @@ void forward_solve_tiled_matrix( const std::size_t m_tile_size, const std::size_t n_tiles, const std::size_t m_tiles, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { for (std::size_t c = 0; c < m_tiles; ++c) { @@ -209,7 +212,7 @@ void backward_solve_tiled_matrix( const std::size_t m_tile_size, const std::size_t n_tiles, const std::size_t m_tiles, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { for (std::size_t c = 0; c < m_tiles; ++c) { @@ -258,7 +261,7 @@ void matrix_vector_tiled(std::vector> &ft_tiles, const std::size_t N_col, const std::size_t n_tiles, const std::size_t m_tiles, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { for (std::size_t k = 0; k < m_tiles; ++k) { @@ -288,7 +291,7 @@ void symmetric_matrix_matrix_diagonal_tiled( const std::size_t m_tile_size, const std::size_t n_tiles, const std::size_t m_tiles, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { for (std::size_t i = 0; i < m_tiles; ++i) { @@ -315,7 +318,7 @@ void compute_gemm_of_invK_y(std::vector> &ft_invK, std::vector> &ft_alpha, const std::size_t n_tile_size, const std::size_t n_tiles, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { for (std::size_t i = 0; i < n_tiles; ++i) { @@ -344,7 +347,7 @@ hpx::shared_future compute_loss_tiled( std::vector> &ft_y, const std::size_t n_tile_size, const std::size_t n_tiles, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { std::vector> loss_tiled(n_tiles); @@ -364,7 +367,7 @@ void symmetric_matrix_matrix_tiled( const std::size_t m_tile_size, const std::size_t n_tiles, const std::size_t m_tiles, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { for (std::size_t c = 0; c < m_tiles; ++c) { @@ -397,7 +400,7 @@ void vector_difference_tiled(std::vector> &ft_prior std::vector> &ft_vector, const std::size_t m_tile_size, const std::size_t m_tiles, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { for (std::size_t i = 0; i < m_tiles; i++) { @@ -409,7 +412,7 @@ void matrix_diagonal_tiled(std::vector> &ft_priorK, std::vector> &ft_vector, const std::size_t m_tile_size, const std::size_t m_tiles, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { for (std::size_t i = 0; i < m_tiles; i++) { @@ -422,7 +425,7 @@ void update_grad_K_tiled_mkl(std::vector> &ft_tiles const std::vector> &ft_v2, const std::size_t n_tile_size, const std::size_t n_tiles, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { for (std::size_t i = 0; i < n_tiles; ++i) { @@ -441,8 +444,8 @@ static double update_hyperparameter( const std::vector> &ft_gradparam, const std::vector> &ft_alpha, double &hyperparameter, // lengthscale or vertical-lengthscale - gprat_hyper::SEKParams sek_params, - gprat_hyper::AdamParams adam_params, + SEKParams sek_params, + AdamParams adam_params, const std::size_t n_tile_size, const std::size_t n_tiles, std::vector> &m_T, @@ -451,7 +454,7 @@ static double update_hyperparameter( const std::vector> &beta2_T, int iter, int param_idx, // 0 for lengthscale, 1 for vertical-lengthscale - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { throw std::logic_error("Function not implemented for GPU"); // return 0; @@ -461,8 +464,8 @@ double update_lengthscale( const std::vector> &ft_invK, const std::vector> &ft_gradparam, const std::vector> &ft_alpha, - gprat_hyper::SEKParams sek_params, - gprat_hyper::AdamParams adam_params, + SEKParams sek_params, + AdamParams adam_params, const std::size_t n_tile_size, const std::size_t n_tiles, std::vector> &m_T, @@ -470,7 +473,7 @@ double update_lengthscale( const std::vector> &beta1_T, const std::vector> &beta2_T, int iter, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { return update_hyperparameter( ft_invK, @@ -494,8 +497,8 @@ double update_vertical_lengthscale( const std::vector> &ft_invK, const std::vector> &ft_gradparam, const std::vector> &ft_alpha, - gprat_hyper::SEKParams sek_params, - gprat_hyper::AdamParams adam_params, + SEKParams sek_params, + AdamParams adam_params, const std::size_t n_tile_size, const std::size_t n_tiles, std::vector> &m_T, @@ -503,7 +506,7 @@ double update_vertical_lengthscale( const std::vector> &beta1_T, const std::vector> &beta2_T, int iter, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { return update_hyperparameter( ft_invK, @@ -526,8 +529,8 @@ double update_vertical_lengthscale( double update_noise_variance( const std::vector> &ft_invK, const std::vector> &ft_alpha, - gprat_hyper::SEKParams sek_params, - gprat_hyper::AdamParams adam_params, + SEKParams sek_params, + AdamParams adam_params, const std::size_t n_tile_size, const std::size_t n_tiles, std::vector> &m_T, @@ -535,10 +538,12 @@ double update_noise_variance( const std::vector> &beta1_T, const std::vector> &beta2_T, int iter, - gprat::CUDA_GPU &gpu) + CUDA_GPU &gpu) { throw std::logic_error("Function not implemented for GPU"); // return 0; } } // end of namespace gpu + +GPRAT_NS_END diff --git a/core/src/target.cpp b/core/src/target.cpp index 1b500702..6b04618f 100644 --- a/core/src/target.cpp +++ b/core/src/target.cpp @@ -1,4 +1,4 @@ -#include "target.hpp" +#include "gprat/target.hpp" #include @@ -7,10 +7,9 @@ using hpx::cuda::experimental::check_cuda_error; #endif -namespace gprat -{ +GPRAT_NS_BEGIN -CPU::CPU() { } +CPU::CPU() = default; bool CPU::is_cpu() { return true; } @@ -154,4 +153,4 @@ int gpu_count() #endif } -} // namespace gprat +GPRAT_NS_END diff --git a/core/src/utils_c.cpp b/core/src/utils_c.cpp index 896b7ad0..664b7808 100644 --- a/core/src/utils_c.cpp +++ b/core/src/utils_c.cpp @@ -1,9 +1,8 @@ -#include "utils_c.hpp" +#include "gprat/utils_c.hpp" #include -namespace utils -{ +GPRAT_NS_BEGIN int compute_train_tiles(int n_samples, int n_tile_size) { @@ -141,4 +140,4 @@ bool compiled_with_cuda() #endif } -} // namespace utils +GPRAT_NS_END diff --git a/examples/gprat_cpp/src/execute.cpp b/examples/gprat_cpp/src/execute.cpp index fa36357a..cd7d13af 100644 --- a/examples/gprat_cpp/src/execute.cpp +++ b/examples/gprat_cpp/src/execute.cpp @@ -1,5 +1,6 @@ #include "gprat/gprat_c.hpp" #include "gprat/utils_c.hpp" + #include #include #include @@ -24,7 +25,7 @@ int main(int argc, char *argv[]) std::string test_path = "../../../data/data_1024/test_input.txt"; bool use_gpu = - utils::compiled_with_cuda() && gprat::gpu_count() > 0 && argc > 1 && std::strcmp(argv[1], "--use_gpu") == 0; + gprat::compiled_with_cuda() && gprat::gpu_count() > 0 && argc > 1 && std::strcmp(argv[1], "--use_gpu") == 0; for (std::size_t core = 2; core <= N_CORES; core = core * 2) { @@ -52,11 +53,11 @@ int main(int argc, char *argv[]) for (std::size_t l = 0; l < LOOP; l++) { // Compute tile sizes and number of predict tiles - int tile_size = utils::compute_train_tile_size(n_train, n_tiles); - auto result = utils::compute_test_tiles(n_test, n_tiles, tile_size); + int tile_size = gprat::compute_train_tile_size(n_train, n_tiles); + auto result = gprat::compute_test_tiles(n_test, n_tiles, tile_size); ///////////////////// ///// hyperparams - gprat_hyper::AdamParams hpar = { 0.1, 0.9, 0.999, 1e-8, OPT_ITER }; + gprat::AdamParams hpar = { 0.1, 0.9, 0.999, 1e-8, OPT_ITER }; ///////////////////// ////// data loading @@ -93,7 +94,7 @@ int main(int argc, char *argv[]) init_time = end_init - start_init; // Initialize HPX with the new arguments, don't run hpx_main - utils::start_hpx_runtime(new_argc, new_argv); + gprat::start_hpx_runtime(new_argc, new_argv); // Measure the time taken to execute gp.cholesky(); auto start_cholesky = std::chrono::high_resolution_clock::now(); @@ -143,7 +144,7 @@ int main(int argc, char *argv[]) init_time = end_init - start_init; // Initialize HPX with the new arguments, don't run hpx_main - utils::start_hpx_runtime(new_argc, new_argv); + gprat::start_hpx_runtime(new_argc, new_argv); auto start_cholesky = std::chrono::high_resolution_clock::now(); std::vector> choleksy_gpu = gp_gpu.cholesky(); @@ -172,7 +173,7 @@ int main(int argc, char *argv[]) } // Stop the HPX runtime - utils::stop_hpx_runtime(); + gprat::stop_hpx_runtime(); auto end_total = std::chrono::high_resolution_clock::now(); auto total_time = end_total - start_total; diff --git a/test/src/output_correctness.cpp b/test/src/output_correctness.cpp index 1c61fdd7..8f17c5ea 100644 --- a/test/src/output_correctness.cpp +++ b/test/src/output_correctness.cpp @@ -1,5 +1,6 @@ #include "gprat/gprat_c.hpp" #include "gprat/utils_c.hpp" + #include #include @@ -73,11 +74,11 @@ gprat_results run_on_data_cpu(const std::string &train_path, const std::string & const std::size_t n_reg = 8; // Compute tile sizes and number of predict tiles - const int tile_size = utils::compute_train_tile_size(n_train, n_tiles); - const auto test_tiles = utils::compute_test_tiles(n_test, n_tiles, tile_size); + const int tile_size = gprat::compute_train_tile_size(n_train, n_tiles); + const auto test_tiles = gprat::compute_test_tiles(n_test, n_tiles, tile_size); // hyperparams - gprat_hyper::AdamParams hpar = { 0.1, 0.9, 0.999, 1e-8, OPT_ITER }; + gprat::AdamParams hpar = { 0.1, 0.9, 0.999, 1e-8, OPT_ITER }; // data loading gprat::GP_data training_input(train_path, n_train, n_reg); @@ -90,7 +91,7 @@ gprat_results run_on_data_cpu(const std::string &train_path, const std::string & training_input.data, training_output.data, n_tiles, tile_size, n_reg, { 1.0, 1.0, 0.1 }, trainable); // Initialize HPX with no arguments, don't run hpx_main - utils::start_hpx_runtime(0, nullptr); + gprat::start_hpx_runtime(0, nullptr); gprat_results results_cpu; @@ -105,7 +106,7 @@ gprat_results run_on_data_cpu(const std::string &train_path, const std::string & results_cpu.pred = gp_cpu.predict(test_input.data, test_tiles.first, test_tiles.second); // Stop the HPX runtime - utils::stop_hpx_runtime(); + gprat::stop_hpx_runtime(); return results_cpu; } @@ -120,8 +121,8 @@ gprat_results run_on_data_gpu(const std::string &train_path, const std::string & const int gpu_id = 0; const int n_streams = 1; - const int tile_size = utils::compute_train_tile_size(n_train, n_tiles); - const auto test_tiles = utils::compute_test_tiles(n_test, n_tiles, tile_size); + const int tile_size = gprat::compute_train_tile_size(n_train, n_tiles); + const auto test_tiles = gprat::compute_test_tiles(n_test, n_tiles, tile_size); gprat::GP_data training_input(train_path, n_train, n_reg); gprat::GP_data training_output(out_path, n_train, n_reg); @@ -139,7 +140,7 @@ gprat_results run_on_data_gpu(const std::string &train_path, const std::string & gpu_id, n_streams); - utils::start_hpx_runtime(0, nullptr); + gprat::start_hpx_runtime(0, nullptr); gprat_results results_gpu; results_gpu.choleksy = gp_gpu.cholesky(); @@ -148,7 +149,7 @@ gprat_results run_on_data_gpu(const std::string &train_path, const std::string & results_gpu.full_no_optimize = gp_gpu.predict_with_full_cov(test_input.data, test_tiles.first, test_tiles.second); results_gpu.pred_no_optimize = gp_gpu.predict(test_input.data, test_tiles.first, test_tiles.second); - utils::stop_hpx_runtime(); + gprat::stop_hpx_runtime(); return results_gpu; } @@ -256,7 +257,7 @@ TEST_CASE("GP CPU results match known-good values", "[integration][cpu]") // NOTE: using higher tolerance than for CPU TEST_CASE("GP GPU results match known-good values (no loss)", "[integration][gpu]") { - if (!utils::compiled_with_cuda()) + if (!gprat::compiled_with_cuda()) { WARN("CUDA not available — skipping GPU test."); return; From 67ce1fcdd8492eb07fecccb493e865e3b9f842cc Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sun, 20 Jul 2025 04:06:43 +0200 Subject: [PATCH 09/56] refactor!(core): Remove unnecessary prefixes/suffixes from filenames Now that our headers are properly namespaced, there's no need to prefix their filenames with gp_ or end them with _c to avoid name clashes with library users. --- bindings/gprat_py.cpp | 2 +- bindings/utils_py.cpp | 2 +- core/CMakeLists.txt | 13 +++++++------ core/include/gprat/cpu/gp_algorithms.hpp | 2 +- core/include/gprat/cpu/gp_functions.hpp | 4 ++-- core/include/gprat/cpu/gp_optimizer.hpp | 4 ++-- core/include/gprat/cpu/tiled_algorithms.hpp | 4 ++-- core/include/gprat/{gprat_c.hpp => gprat.hpp} | 4 ++-- core/include/gprat/gpu/gp_algorithms.cuh | 2 +- core/include/gprat/gpu/gp_functions.cuh | 4 ++-- core/include/gprat/gpu/gp_optimizer.cuh | 4 ++-- core/include/gprat/gpu/tiled_algorithms.cuh | 4 ++-- .../{gp_hyperparameters.hpp => hyperparameters.hpp} | 0 core/include/gprat/{gp_kernels.hpp => kernels.hpp} | 0 core/include/gprat/{utils_c.hpp => utils.hpp} | 4 ++-- core/src/{gprat_c.cpp => gprat.cpp} | 4 ++-- core/src/gpu/gp_algorithms.cu | 2 +- core/src/gpu/gp_functions.cu | 2 +- .../{gp_hyperparameters.cpp => hyperparameters.cpp} | 2 +- core/src/{gp_kernels.cpp => kernels.cpp} | 2 +- core/src/{utils_c.cpp => utils.cpp} | 2 +- examples/gprat_cpp/src/execute.cpp | 4 ++-- test/src/output_correctness.cpp | 4 ++-- 23 files changed, 38 insertions(+), 37 deletions(-) rename core/include/gprat/{gprat_c.hpp => gprat.hpp} (98%) rename core/include/gprat/{gp_hyperparameters.hpp => hyperparameters.hpp} (100%) rename core/include/gprat/{gp_kernels.hpp => kernels.hpp} (100%) rename core/include/gprat/{utils_c.hpp => utils.hpp} (97%) rename core/src/{gprat_c.cpp => gprat.cpp} (99%) rename core/src/{gp_hyperparameters.cpp => hyperparameters.cpp} (94%) rename core/src/{gp_kernels.cpp => kernels.cpp} (97%) rename core/src/{utils_c.cpp => utils.cpp} (99%) diff --git a/bindings/gprat_py.cpp b/bindings/gprat_py.cpp index b122df75..9efb56ce 100644 --- a/bindings/gprat_py.cpp +++ b/bindings/gprat_py.cpp @@ -1,4 +1,4 @@ -#include "gprat/gprat_c.hpp" +#include "gprat/gprat.hpp" #include #include diff --git a/bindings/utils_py.cpp b/bindings/utils_py.cpp index 0fc35506..ab44cc5a 100644 --- a/bindings/utils_py.cpp +++ b/bindings/utils_py.cpp @@ -1,5 +1,5 @@ #include "gprat/target.hpp" -#include "gprat/utils_c.hpp" +#include "gprat/utils.hpp" #include #include diff --git a/core/CMakeLists.txt b/core/CMakeLists.txt index 2ee98d07..c0fe56a2 100644 --- a/core/CMakeLists.txt +++ b/core/CMakeLists.txt @@ -1,18 +1,19 @@ +# Option for GPU support with CUDA, cuSolver, cuBLAS +option(GPRAT_WITH_CUDA "Enable GPU support with CUDA, cuSolver, cuBLAS" OFF) + if(GPRAT_WITH_CUDA) enable_language(CUDA) endif() -# Option for GPU support with CUDA, cuSolver, cuBLAS -option(GPRAT_WITH_CUDA "Enable GPU support with CUDA, cuSolver, cuBLAS" OFF) # Pass variable to C++ code add_compile_definitions(GPRAT_WITH_CUDA=$) set(SOURCE_FILES - src/gprat_c.cpp - src/utils_c.cpp + src/gprat.cpp + src/utils.cpp src/target.cpp - src/gp_kernels.cpp - src/gp_hyperparameters.cpp + src/kernels.cpp + src/hyperparameters.cpp src/cpu/gp_functions.cpp src/cpu/gp_algorithms.cpp src/cpu/gp_uncertainty.cpp diff --git a/core/include/gprat/cpu/gp_algorithms.hpp b/core/include/gprat/cpu/gp_algorithms.hpp index 2ad66542..2285c603 100644 --- a/core/include/gprat/cpu/gp_algorithms.hpp +++ b/core/include/gprat/cpu/gp_algorithms.hpp @@ -4,7 +4,7 @@ #pragma once #include "gprat/detail/config.hpp" -#include "gprat/gp_kernels.hpp" +#include "gprat/kernels.hpp" #include diff --git a/core/include/gprat/cpu/gp_functions.hpp b/core/include/gprat/cpu/gp_functions.hpp index fcd41996..6c5b292f 100644 --- a/core/include/gprat/cpu/gp_functions.hpp +++ b/core/include/gprat/cpu/gp_functions.hpp @@ -4,8 +4,8 @@ #pragma once #include "gprat/detail/config.hpp" -#include "gprat/gp_hyperparameters.hpp" -#include "gprat/gp_kernels.hpp" +#include "gprat/hyperparameters.hpp" +#include "gprat/kernels.hpp" #include diff --git a/core/include/gprat/cpu/gp_optimizer.hpp b/core/include/gprat/cpu/gp_optimizer.hpp index ff9dc1b7..10176faf 100644 --- a/core/include/gprat/cpu/gp_optimizer.hpp +++ b/core/include/gprat/cpu/gp_optimizer.hpp @@ -4,8 +4,8 @@ #pragma once #include "gprat/detail/config.hpp" -#include "gprat/gp_hyperparameters.hpp" -#include "gprat/gp_kernels.hpp" +#include "gprat/hyperparameters.hpp" +#include "gprat/kernels.hpp" #include diff --git a/core/include/gprat/cpu/tiled_algorithms.hpp b/core/include/gprat/cpu/tiled_algorithms.hpp index 56713588..be3593c0 100644 --- a/core/include/gprat/cpu/tiled_algorithms.hpp +++ b/core/include/gprat/cpu/tiled_algorithms.hpp @@ -4,8 +4,8 @@ #pragma once #include "gprat/detail/config.hpp" -#include "gprat/gp_hyperparameters.hpp" -#include "gprat/gp_kernels.hpp" +#include "gprat/hyperparameters.hpp" +#include "gprat/kernels.hpp" #include diff --git a/core/include/gprat/gprat_c.hpp b/core/include/gprat/gprat.hpp similarity index 98% rename from core/include/gprat/gprat_c.hpp rename to core/include/gprat/gprat.hpp index 2596eaed..ab03cd5e 100644 --- a/core/include/gprat/gprat_c.hpp +++ b/core/include/gprat/gprat.hpp @@ -4,8 +4,8 @@ #pragma once #include "gprat/detail/config.hpp" -#include "gprat/gp_hyperparameters.hpp" -#include "gprat/gp_kernels.hpp" +#include "gprat/hyperparameters.hpp" +#include "gprat/kernels.hpp" #include "gprat/target.hpp" #include diff --git a/core/include/gprat/gpu/gp_algorithms.cuh b/core/include/gprat/gpu/gp_algorithms.cuh index 73981b96..d78e1160 100644 --- a/core/include/gprat/gpu/gp_algorithms.cuh +++ b/core/include/gprat/gpu/gp_algorithms.cuh @@ -5,7 +5,7 @@ #include "gprat/detail/config.hpp" -#include "gprat/gp_kernels.hpp" +#include "gprat/kernels.hpp" #include "gprat/target.hpp" #include #include diff --git a/core/include/gprat/gpu/gp_functions.cuh b/core/include/gprat/gpu/gp_functions.cuh index f4949def..780485df 100644 --- a/core/include/gprat/gpu/gp_functions.cuh +++ b/core/include/gprat/gpu/gp_functions.cuh @@ -5,8 +5,8 @@ #include "gprat/detail/config.hpp" -#include "gprat/gp_hyperparameters.hpp" -#include "gprat/gp_kernels.hpp" +#include "gprat/hyperparameters.hpp" +#include "gprat/kernels.hpp" #include "gprat/target.hpp" GPRAT_NS_BEGIN diff --git a/core/include/gprat/gpu/gp_optimizer.cuh b/core/include/gprat/gpu/gp_optimizer.cuh index ebe3aa43..61495de0 100644 --- a/core/include/gprat/gpu/gp_optimizer.cuh +++ b/core/include/gprat/gpu/gp_optimizer.cuh @@ -5,8 +5,8 @@ #include "gprat/detail/config.hpp" -#include "gprat/gp_hyperparameters.hpp" -#include "gprat/gp_kernels.hpp" +#include "gprat/hyperparameters.hpp" +#include "gprat/kernels.hpp" #include "gprat/target.hpp" #include diff --git a/core/include/gprat/gpu/tiled_algorithms.cuh b/core/include/gprat/gpu/tiled_algorithms.cuh index cc850679..38875e1e 100644 --- a/core/include/gprat/gpu/tiled_algorithms.cuh +++ b/core/include/gprat/gpu/tiled_algorithms.cuh @@ -5,9 +5,9 @@ #include "gprat/detail/config.hpp" -#include "gprat/gp_hyperparameters.hpp" +#include "gprat/hyperparameters.hpp" #include "gprat/target.hpp" -#include "gprat/gp_kernels.hpp" +#include "gprat/kernels.hpp" #include #include diff --git a/core/include/gprat/gp_hyperparameters.hpp b/core/include/gprat/hyperparameters.hpp similarity index 100% rename from core/include/gprat/gp_hyperparameters.hpp rename to core/include/gprat/hyperparameters.hpp diff --git a/core/include/gprat/gp_kernels.hpp b/core/include/gprat/kernels.hpp similarity index 100% rename from core/include/gprat/gp_kernels.hpp rename to core/include/gprat/kernels.hpp diff --git a/core/include/gprat/utils_c.hpp b/core/include/gprat/utils.hpp similarity index 97% rename from core/include/gprat/utils_c.hpp rename to core/include/gprat/utils.hpp index 21eb7a72..d269c91c 100644 --- a/core/include/gprat/utils_c.hpp +++ b/core/include/gprat/utils.hpp @@ -1,5 +1,5 @@ -#ifndef GPRAT_UTILS_C_H -#define GPRAT_UTILS_C_H +#ifndef GPRAT_UTILS_HPP +#define GPRAT_UTILS_HPP #pragma once diff --git a/core/src/gprat_c.cpp b/core/src/gprat.cpp similarity index 99% rename from core/src/gprat_c.cpp rename to core/src/gprat.cpp index d7805804..9eb199cf 100644 --- a/core/src/gprat_c.cpp +++ b/core/src/gprat.cpp @@ -1,7 +1,7 @@ -#include "gprat/gprat_c.hpp" +#include "gprat/gprat.hpp" #include "gprat/cpu/gp_functions.hpp" -#include "gprat/utils_c.hpp" +#include "gprat/utils.hpp" #if GPRAT_WITH_CUDA #include "gpu/gp_functions.cuh" diff --git a/core/src/gpu/gp_algorithms.cu b/core/src/gpu/gp_algorithms.cu index 832450fe..5e80df22 100644 --- a/core/src/gpu/gp_algorithms.cu +++ b/core/src/gpu/gp_algorithms.cu @@ -1,9 +1,9 @@ #include "gprat/gpu/gp_algorithms.cuh" -#include "gprat/gp_kernels.hpp" #include "gprat/gpu/cuda_kernels.cuh" #include "gprat/gpu/cuda_utils.cuh" #include "gprat/gpu/gp_optimizer.cuh" +#include "gprat/kernels.hpp" #include "gprat/target.hpp" #include diff --git a/core/src/gpu/gp_functions.cu b/core/src/gpu/gp_functions.cu index 7ba8142b..80d40763 100644 --- a/core/src/gpu/gp_functions.cu +++ b/core/src/gpu/gp_functions.cu @@ -1,9 +1,9 @@ #include "gprat/gpu/gp_functions.cuh" -#include "gprat/gp_kernels.hpp" #include "gprat/gpu/cuda_utils.cuh" #include "gprat/gpu/gp_algorithms.cuh" #include "gprat/gpu/tiled_algorithms.cuh" +#include "gprat/kernels.hpp" #include "gprat/target.hpp" #include diff --git a/core/src/gp_hyperparameters.cpp b/core/src/hyperparameters.cpp similarity index 94% rename from core/src/gp_hyperparameters.cpp rename to core/src/hyperparameters.cpp index a44bdc36..ac355e5c 100644 --- a/core/src/gp_hyperparameters.cpp +++ b/core/src/hyperparameters.cpp @@ -1,4 +1,4 @@ -#include "gprat/gp_hyperparameters.hpp" +#include "gprat/hyperparameters.hpp" #include #include diff --git a/core/src/gp_kernels.cpp b/core/src/kernels.cpp similarity index 97% rename from core/src/gp_kernels.cpp rename to core/src/kernels.cpp index 66b789b4..717bbec6 100644 --- a/core/src/gp_kernels.cpp +++ b/core/src/kernels.cpp @@ -1,4 +1,4 @@ -#include "gprat/gp_kernels.hpp" +#include "gprat/kernels.hpp" #include diff --git a/core/src/utils_c.cpp b/core/src/utils.cpp similarity index 99% rename from core/src/utils_c.cpp rename to core/src/utils.cpp index 664b7808..bbea471b 100644 --- a/core/src/utils_c.cpp +++ b/core/src/utils.cpp @@ -1,4 +1,4 @@ -#include "gprat/utils_c.hpp" +#include "gprat/utils.hpp" #include diff --git a/examples/gprat_cpp/src/execute.cpp b/examples/gprat_cpp/src/execute.cpp index cd7d13af..d9ea2230 100644 --- a/examples/gprat_cpp/src/execute.cpp +++ b/examples/gprat_cpp/src/execute.cpp @@ -1,5 +1,5 @@ -#include "gprat/gprat_c.hpp" -#include "gprat/utils_c.hpp" +#include "gprat/gprat.hpp" +#include "gprat/utils.hpp" #include #include diff --git a/test/src/output_correctness.cpp b/test/src/output_correctness.cpp index 8f17c5ea..250a3341 100644 --- a/test/src/output_correctness.cpp +++ b/test/src/output_correctness.cpp @@ -1,5 +1,5 @@ -#include "gprat/gprat_c.hpp" -#include "gprat/utils_c.hpp" +#include "gprat/gprat.hpp" +#include "gprat/utils.hpp" #include #include From 21b69e672c0c0afa229776bbff96722d5e70e9af Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sun, 20 Jul 2025 21:49:12 +0200 Subject: [PATCH 10/56] fix(ci): Always enable lint workflows They're not costly in terms of workflow minutes so we can just do that. --- .github/workflows/lint.yml | 1 - 1 file changed, 1 deletion(-) diff --git a/.github/workflows/lint.yml b/.github/workflows/lint.yml index ff35d191..c1aee584 100644 --- a/.github/workflows/lint.yml +++ b/.github/workflows/lint.yml @@ -3,7 +3,6 @@ name: Code linting on: push: branches: - - main pull_request: jobs: From 42412bead0dcd155f8142d5ae8f5c136179bc566 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Wed, 23 Jul 2025 01:47:46 +0200 Subject: [PATCH 11/56] feat(core): Support serializing AdamParams and SEKParams --- core/include/gprat/hyperparameters.hpp | 25 +++++++++++++++++++ core/include/gprat/kernels.hpp | 34 +++++++++++++++++++++++--- core/src/kernels.cpp | 8 +++--- 3 files changed, 59 insertions(+), 8 deletions(-) diff --git a/core/include/gprat/hyperparameters.hpp b/core/include/gprat/hyperparameters.hpp index 9ede4756..e81bdf03 100644 --- a/core/include/gprat/hyperparameters.hpp +++ b/core/include/gprat/hyperparameters.hpp @@ -5,6 +5,7 @@ #include "gprat/detail/config.hpp" +#include #include GPRAT_NS_BEGIN @@ -58,6 +59,30 @@ struct AdamParams std::string repr() const; }; +template +void save_construct_data(Archive &ar, const AdamParams *v, const unsigned int) +{ + ar << v->learning_rate; + ar << v->beta1; + ar << v->beta2; + ar << v->epsilon; + ar << v->opt_iter; +} + +template +void load_construct_data(Archive &ar, AdamParams *v, const unsigned int) +{ + double learning_rate, beta1, beta2, epsilon; + int opt_iter; + ar >> learning_rate; + ar >> beta1; + ar >> beta2; + ar >> epsilon; + ar >> opt_iter; + + std::construct_at(v, learning_rate, beta1, beta2, epsilon, opt_iter); +} + GPRAT_NS_END #endif diff --git a/core/include/gprat/kernels.hpp b/core/include/gprat/kernels.hpp index 8b5dc9c1..0b489089 100644 --- a/core/include/gprat/kernels.hpp +++ b/core/include/gprat/kernels.hpp @@ -6,6 +6,7 @@ #include "gprat/detail/config.hpp" #include +#include #include GPRAT_NS_BEGIN @@ -43,12 +44,12 @@ struct SEKParams /** * @brief Construct a new SEKParams object * - * @param lengthscale Lengthscale: variance of training output - * @param vertical_lengthscale Vertical Lengthscale: standard deviation + * @param in_lengthscale Lengthscale: variance of training output + * @param in_vertical_lengthscale Vertical Lengthscale: standard deviation * of training input - * @param noise_variance Noise Variance: small value + * @param in_noise_variance Noise Variance: small value */ - SEKParams(double lengthscale_, double vertical_lengthscale_, double noise_variance_); + SEKParams(double in_lengthscale, double in_vertical_lengthscale, double in_noise_variance); /** * @brief Return the number of parameters @@ -79,6 +80,31 @@ struct SEKParams const double &get_param(std::size_t index) const; }; +template +void save_construct_data(Archive &ar, const SEKParams *v, const unsigned int) +{ + ar << v->lengthscale; + ar << v->vertical_lengthscale; + ar << v->noise_variance; +} + +template +void load_construct_data(Archive &ar, SEKParams *v, const unsigned int) +{ + double lengthscale, vertical_lengthscale, noise_variance; + ar >> lengthscale; + ar >> vertical_lengthscale; + ar >> noise_variance; + + std::construct_at(v, lengthscale, vertical_lengthscale, noise_variance); +} + +template +void serialize(Archive &ar, SEKParams &pt, const unsigned int) +{ + ar & pt.m_T & pt.w_T; +} + GPRAT_NS_END #endif diff --git a/core/src/kernels.cpp b/core/src/kernels.cpp index 717bbec6..9fd0218e 100644 --- a/core/src/kernels.cpp +++ b/core/src/kernels.cpp @@ -4,10 +4,10 @@ GPRAT_NS_BEGIN -SEKParams::SEKParams(double lengthscale, double vertical_lengthscale, double noise_variance) : - lengthscale(lengthscale), - vertical_lengthscale(vertical_lengthscale), - noise_variance(noise_variance) +SEKParams::SEKParams(double in_lengthscale, double in_vertical_lengthscale, double in_noise_variance) : + lengthscale(in_lengthscale), + vertical_lengthscale(in_vertical_lengthscale), + noise_variance(in_noise_variance) { m_T.resize(this->size()); w_T.resize(this->size()); From 5ce8c1a459466ba5944c8dc0e250da0a997c0bd1 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sun, 20 Jul 2025 05:58:13 +0200 Subject: [PATCH 12/56] feat!(core): Introduce const_tile_data + mutable_tile_data in lieu of std::vector for tiles of type T. The advantage of this is: - tiles are easily HPX-serializable and we can put them into HPX components - we can perhaps later add support for automatic GPU upload --- core/CMakeLists.txt | 1 + core/include/gprat/cpu/adapter_cblas_fp32.hpp | 96 +++--- core/include/gprat/cpu/adapter_cblas_fp64.hpp | 106 +++--- core/include/gprat/cpu/gp_algorithms.hpp | 42 ++- core/include/gprat/cpu/gp_functions.hpp | 17 +- core/include/gprat/cpu/gp_optimizer.hpp | 35 +- core/include/gprat/cpu/gp_uncertainty.hpp | 7 +- core/include/gprat/cpu/tiled_algorithms.hpp | 5 +- core/include/gprat/detail/async_helpers.hpp | 33 ++ core/include/gprat/gprat.hpp | 3 +- core/include/gprat/performance_counters.hpp | 20 ++ core/include/gprat/tile_data.hpp | 133 ++++++++ core/src/cpu/adapter_cblas_fp32.cpp | 105 +++--- core/src/cpu/adapter_cblas_fp64.cpp | 104 +++--- core/src/cpu/gp_algorithms.cpp | 129 ++++---- core/src/cpu/gp_functions.cpp | 301 ++++++------------ core/src/cpu/gp_optimizer.cpp | 74 ++--- core/src/cpu/gp_uncertainty.cpp | 15 +- core/src/cpu/tiled_algorithms.cpp | 137 +++----- core/src/gprat.cpp | 2 +- core/src/performance_counters.cpp | 37 +++ examples/gprat_cpp/src/execute.cpp | 22 +- test/src/output_correctness.cpp | 39 ++- 23 files changed, 765 insertions(+), 698 deletions(-) create mode 100644 core/include/gprat/detail/async_helpers.hpp create mode 100644 core/include/gprat/performance_counters.hpp create mode 100644 core/include/gprat/tile_data.hpp create mode 100644 core/src/performance_counters.cpp diff --git a/core/CMakeLists.txt b/core/CMakeLists.txt index c0fe56a2..8309eba2 100644 --- a/core/CMakeLists.txt +++ b/core/CMakeLists.txt @@ -11,6 +11,7 @@ add_compile_definitions(GPRAT_WITH_CUDA=$) set(SOURCE_FILES src/gprat.cpp src/utils.cpp + src/performance_counters.cpp src/target.cpp src/kernels.cpp src/hyperparameters.cpp diff --git a/core/include/gprat/cpu/adapter_cblas_fp32.hpp b/core/include/gprat/cpu/adapter_cblas_fp32.hpp index fa6272b5..54852fee 100644 --- a/core/include/gprat/cpu/adapter_cblas_fp32.hpp +++ b/core/include/gprat/cpu/adapter_cblas_fp32.hpp @@ -4,14 +4,13 @@ #pragma once #include "gprat/detail/config.hpp" +#include "gprat/tile_data.hpp" #include #include GPRAT_NS_BEGIN -using vector_future = hpx::shared_future>; - // Constants that are compatible with CBLAS typedef enum BLAS_TRANSPOSE { Blas_no_trans = 111, Blas_trans = 112 } BLAS_TRANSPOSE; @@ -29,41 +28,42 @@ typedef enum BLAS_ALPHA { Blas_add = 1, Blas_substract = -1 } BLAS_ALPHA; /** * @brief FP32 In-place Cholesky decomposition of A - * @param f_A matrix to be factorized + * @param A matrix to be factorized * @param N matrix dimension * @return factorized, lower triangular matrix f_L */ -vector_future potrf(vector_future f_A, const int N); +mutable_tile_data potrf(const mutable_tile_data &A, int N); /** * @brief FP32 In-place solve L(^T) * X = A or X * L(^T) = A where L lower triangular - * @param f_L Cholesky factor matrix - * @param f_A right hand side matrix + * @param L Cholesky factor matrix + * @param A right hand side matrix * @param N first dimension * @param M second dimension * @return solution matrix f_X */ -vector_future trsm(vector_future f_L, - vector_future f_A, - const int N, - const int M, - const BLAS_TRANSPOSE transpose_L, - const BLAS_SIDE side_L); +mutable_tile_data +trsm(const const_tile_data &L, + const mutable_tile_data &A, + int N, + int M, + BLAS_TRANSPOSE transpose_L, + BLAS_SIDE side_L); /** * @brief FP32 Symmetric rank-k update: A = A - B * B^T - * @param f_A Base matrix - * @param f_B Symmetric update matrix + * @param A Base matrix + * @param B Symmetric update matrix * @param N matrix dimension * @return updated matrix f_A */ -vector_future syrk(vector_future f_A, vector_future f_B, const int N); +mutable_tile_data syrk(const mutable_tile_data &A, const const_tile_data &B, int N); /** * @brief FP32 General matrix-matrix multiplication: C = C - A(^T) * B(^T) - * @param f_C Base matrix - * @param f_B Right update matrix - * @param f_A Left update matrix + * @param C Base matrix + * @param B Right update matrix + * @param A Left update matrix * @param N first matrix dimension * @param M second matrix dimension * @param K third matrix dimension @@ -71,27 +71,28 @@ vector_future syrk(vector_future f_A, vector_future f_B, const int N); * @param transpose_B transpose right matrix * @return updated matrix f_X */ -vector_future -gemm(vector_future f_A, - vector_future f_B, - vector_future f_C, - const int N, - const int M, - const int K, - const BLAS_TRANSPOSE transpose_A, - const BLAS_TRANSPOSE transpose_B); +mutable_tile_data +gemm(const const_tile_data &A, + const const_tile_data &B, + const mutable_tile_data &C, + int N, + int M, + int K, + BLAS_TRANSPOSE transpose_A, + BLAS_TRANSPOSE transpose_B); // BLAS level 2 operations /** * @brief FP32 In-place solve L(^T) * x = a where L lower triangular - * @param f_L Cholesky factor matrix - * @param f_a right hand side vector + * @param L Cholesky factor matrix + * @param a right hand side vector * @param N matrix dimension * @param transpose_L transpose Cholesky factor * @return solution vector f_x */ -vector_future trsv(vector_future f_L, vector_future f_a, const int N, const BLAS_TRANSPOSE transpose_L); +mutable_tile_data +trsv(const const_tile_data &L, const mutable_tile_data &a, int N, BLAS_TRANSPOSE transpose_L); /** * @brief FP32 General matrix-vector multiplication: b = b - A(^T) * a @@ -103,34 +104,37 @@ vector_future trsv(vector_future f_L, vector_future f_a, const int N, const BLAS * @param transpose_A transpose update matrix * @return updated vector f_b */ -vector_future gemv(vector_future f_A, - vector_future f_a, - vector_future f_b, - const int N, - const int M, - const BLAS_ALPHA alpha, - const BLAS_TRANSPOSE transpose_A); +mutable_tile_data +gemv(const const_tile_data &A, + const const_tile_data &a, + const mutable_tile_data &b, + int N, + int M, + BLAS_ALPHA alpha, + BLAS_TRANSPOSE transpose_A); /** * @brief FP32 Vector update with diagonal SYRK: r = r + diag(A^T * A) - * @param f_A update matrix - * @param f_r base vector + * @param A update matrix + * @param r base vector * @param N first matrix dimension * @param M second matrix dimension * @return updated vector f_r */ -vector_future dot_diag_syrk(vector_future f_A, vector_future f_r, const int N, const int M); +mutable_tile_data +dot_diag_syrk(const const_tile_data &A, const mutable_tile_data &r, int N, int M); /** * @brief FP32 Vector update with diagonal GEMM: r = r + diag(A * B) - * @param f_A first update matrix - * @param f_B second update matrix - * @param f_r base vector + * @param A first update matrix + * @param B second update matrix + * @param r base vector * @param N first matrix dimension * @param M second matrix dimension * @return updated vector f_r */ -vector_future dot_diag_gemm(vector_future f_A, vector_future f_B, vector_future f_r, const int N, const int M); +mutable_tile_data dot_diag_gemm( + const const_tile_data &A, const const_tile_data &B, const mutable_tile_data &r, int N, int M); // BLAS level 1 operations @@ -141,7 +145,7 @@ vector_future dot_diag_gemm(vector_future f_A, vector_future f_B, vector_future * @param N vector length * @return y - x */ -vector_future axpy(vector_future f_y, vector_future f_x, const int N); +mutable_tile_data axpy(const mutable_tile_data &y, const const_tile_data &x, int N); /** * @brief FP32 Dot product: a * b @@ -150,7 +154,7 @@ vector_future axpy(vector_future f_y, vector_future f_x, const int N); * @param N vector length * @return f_a * f_b */ -float dot(std::vector a, std::vector b, const int N); +float dot(std::span a, std::span b, int N); GPRAT_NS_END diff --git a/core/include/gprat/cpu/adapter_cblas_fp64.hpp b/core/include/gprat/cpu/adapter_cblas_fp64.hpp index fbf81b9d..4527c8bd 100644 --- a/core/include/gprat/cpu/adapter_cblas_fp64.hpp +++ b/core/include/gprat/cpu/adapter_cblas_fp64.hpp @@ -4,14 +4,13 @@ #pragma once #include "gprat/detail/config.hpp" +#include "gprat/tile_data.hpp" #include #include GPRAT_NS_BEGIN -using vector_future = hpx::shared_future>; - // Constants that are compatible with CBLAS typedef enum BLAS_TRANSPOSE { Blas_no_trans = 111, Blas_trans = 112 } BLAS_TRANSPOSE; @@ -29,41 +28,42 @@ typedef enum BLAS_ALPHA { Blas_add = 1, Blas_substract = -1 } BLAS_ALPHA; /** * @brief FP64 In-place Cholesky decomposition of A - * @param f_A matrix to be factorized + * @param A matrix to be factorized * @param N matrix dimension * @return factorized, lower triangular matrix f_L */ -vector_future potrf(vector_future f_A, const int N); +mutable_tile_data potrf(const mutable_tile_data &A, int N); /** * @brief FP64 In-place solve L(^T) * X = A or X * L(^T) = A where L lower triangular - * @param f_L Cholesky factor matrix - * @param f_A right hand side matrix + * @param L Cholesky factor matrix + * @param A right hand side matrix * @param N first dimension * @param M second dimension * @return solution matrix f_X */ -vector_future trsm(vector_future f_L, - vector_future f_A, - const int N, - const int M, - const BLAS_TRANSPOSE transpose_L, - const BLAS_SIDE side_L); +mutable_tile_data +trsm(const const_tile_data &L, + const mutable_tile_data &A, + int N, + int M, + BLAS_TRANSPOSE transpose_L, + BLAS_SIDE side_L); /** * @brief FP64 Symmetric rank-k update: A = A - B * B^T - * @param f_A Base matrix - * @param f_B Symmetric update matrix + * @param A Base matrix + * @param B Symmetric update matrix * @param N matrix dimension * @return updated matrix f_A */ -vector_future syrk(vector_future f_A, vector_future f_B, const int N); +mutable_tile_data syrk(const mutable_tile_data &A, const const_tile_data &B, int N); /** * @brief FP64 General matrix-matrix multiplication: C = C - A(^T) * B(^T) - * @param f_C Base matrix - * @param f_B Right update matrix - * @param f_A Left update matrix + * @param C Base matrix + * @param B Right update matrix + * @param A Left update matrix * @param N first matrix dimension * @param M second matrix dimension * @param K third matrix dimension @@ -71,66 +71,74 @@ vector_future syrk(vector_future f_A, vector_future f_B, const int N); * @param transpose_B transpose right matrix * @return updated matrix f_X */ -vector_future -gemm(vector_future f_A, - vector_future f_B, - vector_future f_C, - const int N, - const int M, - const int K, - const BLAS_TRANSPOSE transpose_A, - const BLAS_TRANSPOSE transpose_B); +mutable_tile_data +gemm(const const_tile_data &A, + const const_tile_data &B, + const mutable_tile_data &C, + int N, + int M, + int K, + BLAS_TRANSPOSE transpose_A, + BLAS_TRANSPOSE transpose_B); // BLAS level 2 operations /** * @brief FP64 In-place solve L(^T) * x = a where L lower triangular - * @param f_L Cholesky factor matrix - * @param f_a right hand side vector + * @param L Cholesky factor matrix + * @param a right hand side vector * @param N matrix dimension * @param transpose_L transpose Cholesky factor * @return solution vector f_x */ -vector_future trsv(vector_future f_L, vector_future f_a, const int N, const BLAS_TRANSPOSE transpose_L); +mutable_tile_data +trsv(const const_tile_data &L, const mutable_tile_data &a, int N, BLAS_TRANSPOSE transpose_L); /** * @brief FP64 General matrix-vector multiplication: b = b - A(^T) * a - * @param f_A update matrix - * @param f_a update vector - * @param f_b base vector + * @param A update matrix + * @param a update vector + * @param b base vector * @param N matrix dimension * @param alpha add or substract update to base vector * @param transpose_A transpose update matrix * @return updated vector f_b */ -vector_future gemv(vector_future f_A, - vector_future f_a, - vector_future f_b, - const int N, - const int M, - const BLAS_ALPHA alpha, - const BLAS_TRANSPOSE transpose_A); +mutable_tile_data +gemv(const const_tile_data &A, + const const_tile_data &a, + const mutable_tile_data &b, + int N, + int M, + BLAS_ALPHA alpha, + BLAS_TRANSPOSE transpose_A); /** * @brief FP64 Vector update with diagonal SYRK: r = r + diag(A^T * A) - * @param f_A update matrix - * @param f_r base vector + * @param A update matrix + * @param r base vector * @param N first matrix dimension * @param M second matrix dimension * @return updated vector f_r */ -vector_future dot_diag_syrk(vector_future f_A, vector_future f_r, const int N, const int M); +mutable_tile_data +dot_diag_syrk(const const_tile_data &A, const mutable_tile_data &r, int N, int M); /** * @brief FP64 Vector update with diagonal GEMM: r = r + diag(A * B) - * @param f_A first update matrix - * @param f_B second update matrix - * @param f_r base vector + * @param A first update matrix + * @param B second update matrix + * @param r base vector * @param N first matrix dimension * @param M second matrix dimension * @return updated vector f_r */ -vector_future dot_diag_gemm(vector_future f_A, vector_future f_B, vector_future f_r, const int N, const int M); +mutable_tile_data +dot_diag_gemm(const const_tile_data &A, + const const_tile_data &B, + const mutable_tile_data &r, + int N, + int M); // BLAS level 1 operations @@ -141,7 +149,7 @@ vector_future dot_diag_gemm(vector_future f_A, vector_future f_B, vector_future * @param N vector length * @return y - x */ -vector_future axpy(vector_future f_y, vector_future f_x, const int N); +mutable_tile_data axpy(const mutable_tile_data &y, const const_tile_data &x, int N); /** * @brief FP64 Dot product: a * b @@ -150,7 +158,7 @@ vector_future axpy(vector_future f_y, vector_future f_x, const int N); * @param N vector length * @return a * b */ -double dot(std::vector a, std::vector b, const int N); +double dot(std::span a, std::span b, int N); GPRAT_NS_END diff --git a/core/include/gprat/cpu/gp_algorithms.hpp b/core/include/gprat/cpu/gp_algorithms.hpp index 2285c603..210810fd 100644 --- a/core/include/gprat/cpu/gp_algorithms.hpp +++ b/core/include/gprat/cpu/gp_algorithms.hpp @@ -5,7 +5,9 @@ #include "gprat/detail/config.hpp" #include "gprat/kernels.hpp" +#include "gprat/tile_data.hpp" +#include #include GPRAT_NS_BEGIN @@ -16,21 +18,17 @@ namespace cpu /** * @brief Compute the squared exponential kernel of two feature vectors * - * @param i_global The global index of the first feature vector - * @param j_global The global index of the second feature vector * @param n_regressors The number of regressors - * @param hyperparameters The kernel hyperparameters + * @param sek_params The kernel hyperparameters * @param i_input The first feature vector * @param j_input The second feature vector * - * @return The entry of a covariance function at position i_global,j_global + * @return The entry of a covariance function */ -double compute_covariance_function(std::size_t i_global, - std::size_t j_global, - std::size_t n_regressors, +double compute_covariance_function(std::size_t n_regressors, const SEKParams &sek_params, - const std::vector &i_input, - const std::vector &j_input); + std::span i_input, + std::span j_input); /** * @brief Generate a tile of the covariance matrix @@ -45,13 +43,13 @@ double compute_covariance_function(std::size_t i_global, * @return A quadratic tile of the covariance matrix of size N x N * @note Does apply noise variance on the diagonal */ -std::vector gen_tile_covariance( +mutable_tile_data gen_tile_covariance( std::size_t row, std::size_t col, std::size_t N, std::size_t n_regressors, const SEKParams &sek_params, - const std::vector &input); + std::span input); /** * @brief Generate a tile of the prior covariance matrix @@ -67,13 +65,13 @@ std::vector gen_tile_covariance( * @note Does NOT apply noise variance on the diagonal */ // NAME: gen_tile_priot_covariance -std::vector gen_tile_full_prior_covariance( +mutable_tile_data gen_tile_full_prior_covariance( std::size_t row, std::size_t col, std::size_t N, std::size_t n_regressors, const SEKParams &sek_params, - const std::vector &input); + std::span input); /** * @brief Generate the diagonal of a diagonal tile in the prior covariance matrix @@ -89,13 +87,13 @@ std::vector gen_tile_full_prior_covariance( * @note Does NOT apply noise variance */ // NAME: gen_tile_diag_prior_covariance -std::vector gen_tile_prior_covariance( +mutable_tile_data gen_tile_prior_covariance( std::size_t row, std::size_t col, std::size_t N, std::size_t n_regressors, const SEKParams &sek_params, - const std::vector &input); + std::span input); /** * @brief Generate a tile of the cross-covariance matrix @@ -111,15 +109,15 @@ std::vector gen_tile_prior_covariance( * @return A tile of the cross covariance matrix of size N_row x N_col * @note Does NOT apply noise variance */ -std::vector gen_tile_cross_covariance( +mutable_tile_data gen_tile_cross_covariance( std::size_t row, std::size_t col, std::size_t N_row, std::size_t N_col, std::size_t n_regressors, const SEKParams &sek_params, - const std::vector &row_input, - const std::vector &col_input); + std::span row_input, + std::span col_input); /** * @brief Transpose a tile of size N_row x N_col @@ -130,7 +128,7 @@ std::vector gen_tile_cross_covariance( * * @return The transposed tile of size N_col x N_row */ -std::vector gen_tile_transpose(std::size_t N_row, std::size_t N_col, const std::vector &tile); +mutable_tile_data gen_tile_transpose(std::size_t N_row, std::size_t N_col, std::span tile); /** * @brief Generate a tile of the output data @@ -141,7 +139,7 @@ std::vector gen_tile_transpose(std::size_t N_row, std::size_t N_col, con * * @return A tile of the output data of size N */ -std::vector gen_tile_output(std::size_t row, std::size_t N, const std::vector &output); +mutable_tile_data gen_tile_output(std::size_t row, std::size_t N, std::span output); /** * @brief Compute the L2-error norm over all tiles and elements @@ -164,7 +162,7 @@ double compute_error_norm(std::size_t n_tiles, * * @return A tile filled with zeros of size N */ -std::vector gen_tile_zeros(std::size_t N); +mutable_tile_data gen_tile_zeros(std::size_t N); /** * @brief Generate an identity tile (i==j?1:0) @@ -172,7 +170,7 @@ std::vector gen_tile_zeros(std::size_t N); * @param N The dimension of the quadratic tile * @return A NxN identity tile */ -std::vector gen_tile_identity(std::size_t N); +mutable_tile_data gen_tile_identity(std::size_t N); } // end of namespace cpu diff --git a/core/include/gprat/cpu/gp_functions.hpp b/core/include/gprat/cpu/gp_functions.hpp index 6c5b292f..11a61617 100644 --- a/core/include/gprat/cpu/gp_functions.hpp +++ b/core/include/gprat/cpu/gp_functions.hpp @@ -6,6 +6,7 @@ #include "gprat/detail/config.hpp" #include "gprat/hyperparameters.hpp" #include "gprat/kernels.hpp" +#include "gprat/tile_data.hpp" #include @@ -26,7 +27,7 @@ namespace cpu * * @return The tiled Cholesky factor */ -std::vector> +std::vector> cholesky(const std::vector &training_input, const SEKParams &sek_params, int n_tiles, @@ -149,9 +150,9 @@ double compute_loss(const std::vector &training_input, std::vector optimize(const std::vector &training_input, const std::vector &training_output, - int n_tiles, - int n_tile_size, - int n_regressors, + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors, const AdamParams &adam_params, SEKParams &sek_params, std::vector trainable_params); @@ -176,13 +177,13 @@ optimize(const std::vector &training_input, */ double optimize_step(const std::vector &training_input, const std::vector &training_output, - int n_tiles, - int n_tile_size, - int n_regressors, + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors, AdamParams &adam_params, SEKParams &sek_params, std::vector trainable_params, - int iter); + std::size_t iter); } // end of namespace cpu diff --git a/core/include/gprat/cpu/gp_optimizer.hpp b/core/include/gprat/cpu/gp_optimizer.hpp index 10176faf..1712597d 100644 --- a/core/include/gprat/cpu/gp_optimizer.hpp +++ b/core/include/gprat/cpu/gp_optimizer.hpp @@ -6,6 +6,7 @@ #include "gprat/detail/config.hpp" #include "gprat/hyperparameters.hpp" #include "gprat/kernels.hpp" +#include "gprat/tile_data.hpp" #include @@ -76,7 +77,7 @@ double compute_covariance_distance(std::size_t i_global, * * @return A quadratic tile containing the distance between the features of size N x N */ -std::vector gen_tile_distance( +mutable_tile_data gen_tile_distance( std::size_t row, std::size_t col, std::size_t N, @@ -95,8 +96,12 @@ std::vector gen_tile_distance( * * @return A quadratic tile of the covariance matrix of size N x N */ -std::vector gen_tile_covariance_with_distance( - std::size_t row, std::size_t col, std::size_t N, const SEKParams &sek_params, const std::vector &distance); +mutable_tile_data gen_tile_covariance_with_distance( + std::size_t row, + std::size_t col, + std::size_t N, + const SEKParams &sek_params, + const const_tile_data &distance); /** * @brief Generate a derivative tile w.r.t. vertical_lengthscale v @@ -107,7 +112,8 @@ std::vector gen_tile_covariance_with_distance( * * @return A quadratic tile of the derivative of v of size N x N */ -std::vector gen_tile_grad_v(std::size_t N, const SEKParams &sek_params, const std::vector &distance); +mutable_tile_data +gen_tile_grad_v(std::size_t N, const SEKParams &sek_params, const const_tile_data &distance); /** * @brief Generate a derivative tile w.r.t. lengthscale l @@ -118,7 +124,8 @@ std::vector gen_tile_grad_v(std::size_t N, const SEKParams &sek_params, * * @return A quadratic tile of the derivative of l of size N x N */ -std::vector gen_tile_grad_l(std::size_t N, const SEKParams &sek_params, const std::vector &distance); +mutable_tile_data +gen_tile_grad_l(std::size_t N, const SEKParams &sek_params, const const_tile_data &distance); /** * @brief Update biased first raw moment estimate: m_T+1 = beta_1 * m_T + (1 - beta_1) * g_T. @@ -153,8 +160,8 @@ double update_second_moment(double gradient, double v_T, double beta_2); * * @return The updated hyperparameter */ -double adam_step( - const double unconstrained_hyperparam, const AdamParams &adam_params, double m_T, double v_T, std::size_t iter); +double +adam_step(double unconstrained_hyperparam, const AdamParams &adam_params, double m_T, double v_T, std::size_t iter); /** * @brief Compute negative-log likelihood on one tile. @@ -165,9 +172,9 @@ double adam_step( * * @return Return l = y^T * alpha + \sum_i^N log(L_ii^2) */ -double compute_loss(const std::vector &K_diag_tile, - const std::vector &alpha_tile, - const std::vector &y_tile, +double compute_loss(std::span K_diag_tile, + std::span alpha_tile, + std::span y_tile, std::size_t N); /** @@ -179,7 +186,7 @@ double compute_loss(const std::vector &K_diag_tile, * * @return The added up loss plus the constant factor */ -double add_losses(const std::vector &losses, std::size_t N, std::size_t n); +double add_losses(std::span losses, std::size_t N, std::size_t n); /** * @brief Compute the loss gradient. @@ -201,7 +208,7 @@ double compute_gradient(double trace, double dot, std::size_t N, std::size_t n_t * * @return The updated global trace */ -double compute_trace(const std::vector &diagonal, double trace); +double compute_trace(std::span diagonal, double trace); /** * @brief Add the dot product of a vector to a global result. @@ -212,7 +219,7 @@ double compute_trace(const std::vector &diagonal, double trace); * * @return The updated global result */ -double compute_dot(const std::vector &vector_T, const std::vector &vector, double result); +double compute_dot(std::span vector_T, std::span vector, double result); /** * @brief Add the local trace of a matrix tile to the global trace @@ -223,7 +230,7 @@ double compute_dot(const std::vector &vector_T, const std::vector &tile, double trace, std::size_t N); +double compute_trace_diag(std::span tile, double trace, std::size_t N); } // end of namespace cpu diff --git a/core/include/gprat/cpu/gp_uncertainty.hpp b/core/include/gprat/cpu/gp_uncertainty.hpp index 705f7798..cb402119 100644 --- a/core/include/gprat/cpu/gp_uncertainty.hpp +++ b/core/include/gprat/cpu/gp_uncertainty.hpp @@ -4,9 +4,7 @@ #pragma once #include "gprat/detail/config.hpp" - -#include -#include +#include "gprat/tile_data.hpp" GPRAT_NS_BEGIN @@ -21,8 +19,7 @@ namespace cpu * * @return Diagonal element vector of the matrix A of size M */ -// std::vector get_matrix_diagonal(const std::vector &A, std::size_t M); -hpx::shared_future> get_matrix_diagonal(hpx::shared_future> f_A, std::size_t M); +mutable_tile_data get_matrix_diagonal(const const_tile_data &A, std::size_t M); } // end of namespace cpu diff --git a/core/include/gprat/cpu/tiled_algorithms.hpp b/core/include/gprat/cpu/tiled_algorithms.hpp index be3593c0..0f297b1b 100644 --- a/core/include/gprat/cpu/tiled_algorithms.hpp +++ b/core/include/gprat/cpu/tiled_algorithms.hpp @@ -6,13 +6,14 @@ #include "gprat/detail/config.hpp" #include "gprat/hyperparameters.hpp" #include "gprat/kernels.hpp" +#include "gprat/tile_data.hpp" #include GPRAT_NS_BEGIN -using Tiled_matrix = std::vector>>; -using Tiled_vector = std::vector>>; +using Tiled_matrix = std::vector>>; +using Tiled_vector = std::vector>>; namespace cpu { diff --git a/core/include/gprat/detail/async_helpers.hpp b/core/include/gprat/detail/async_helpers.hpp new file mode 100644 index 00000000..b04ef144 --- /dev/null +++ b/core/include/gprat/detail/async_helpers.hpp @@ -0,0 +1,33 @@ +#ifndef GPRAT_DETAIL_DATAFLOW_HELPERS_HPP +#define GPRAT_DETAIL_DATAFLOW_HELPERS_HPP + +#pragma once + +#include "gprat/detail/config.hpp" + +#include +#include +#include + +GPRAT_NS_BEGIN + +namespace detail +{ + +template +decltype(auto) named_dataflow(const char *name, Args &&...args) +{ + return hpx::dataflow(hpx::annotated_function(hpx::unwrapping(F), name), std::forward(args)...); +} + +template +decltype(auto) named_async(const char *name, Args &&...args) +{ + return hpx::async(hpx::annotated_function(F, name), std::forward(args)...); +} + +} // namespace detail + +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/gprat.hpp b/core/include/gprat/gprat.hpp index ab03cd5e..e41850b2 100644 --- a/core/include/gprat/gprat.hpp +++ b/core/include/gprat/gprat.hpp @@ -8,6 +8,7 @@ #include "gprat/kernels.hpp" #include "gprat/target.hpp" +#include "tile_data.hpp" #include #include #include @@ -226,7 +227,7 @@ class GP /** * @brief Computes & returns cholesky decomposition */ - std::vector> cholesky(); + std::vector> cholesky(); }; GPRAT_NS_END diff --git a/core/include/gprat/performance_counters.hpp b/core/include/gprat/performance_counters.hpp new file mode 100644 index 00000000..86c35c82 --- /dev/null +++ b/core/include/gprat/performance_counters.hpp @@ -0,0 +1,20 @@ +#ifndef GPRAT_PERFORMANCE_COUNTERS_HPP +#define GPRAT_PERFORMANCE_COUNTERS_HPP + +#pragma once + +#include "gprat/detail/config.hpp" + +#include +#include + +GPRAT_NS_BEGIN + +void track_tile_data_allocation(std::size_t size); +void track_tile_data_deallocation(std::size_t size); + +void register_performance_counters(); + +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/tile_data.hpp b/core/include/gprat/tile_data.hpp new file mode 100644 index 00000000..39d48dd9 --- /dev/null +++ b/core/include/gprat/tile_data.hpp @@ -0,0 +1,133 @@ +#ifndef GPRAT_TILE_DATA_HPP +#define GPRAT_TILE_DATA_HPP + +#pragma once + +#include "gprat/detail/config.hpp" +#include "gprat/performance_counters.hpp" + +#include +#include + +GPRAT_NS_BEGIN + +/** + * @brief Non-mutable reference-counted dynamic array of a given type T. + * This class represents a simple reference-counted non-resizeable buffer with elements of type T. + * It can be serialized by HPX and thus be used as a parameter for HPX actions. + * This type is intended to be used for parameters and attributes that do not require mutable data (i.e., only read + * access) + * + * @tparam T Element type of the tile. Usually some numeric type like double or float. This class currently only + * requires T to be serializable by HPX. + */ +template +class const_tile_data +{ + protected: + typedef hpx::serialization::serialize_buffer cpu_buffer_type; + + struct hold_reference + { + explicit hold_reference(const cpu_buffer_type &data) : + data_(data) + { } + + void operator()(const T *) const { } // no deletion necessary + + cpu_buffer_type data_; + }; + + // In case we want pooling down the road... + static T *allocate(std::size_t n) + { + track_tile_data_allocation(n); + return new T[n]; + } + + static void deallocate(T *p) noexcept + { + track_tile_data_deallocation(0); // we don't know here + delete[] p; + } + + public: + const_tile_data() = default; + + // Create a new (uninitialized) tile_data of the given size. + explicit const_tile_data(std::size_t size) : + cpu_data_(allocate(size), size, cpu_buffer_type::take, &const_tile_data::deallocate) + { } + + // Create a tile_data which acts as a proxy to a part of the embedded array. + // The proxy is assumed to refer to either the left or the right boundary + // element. + const_tile_data(const const_tile_data &base, std::size_t offset, std::size_t size) : + cpu_data_(base.cpu_data_.data() + offset, + size, + cpu_buffer_type::reference, + hold_reference(base.cpu_data_)) // keep referenced tile_data alive + { } + + [[nodiscard]] const T *data() const noexcept { return cpu_data_.data(); } + + [[nodiscard]] std::size_t size() const noexcept { return cpu_data_.size(); } + + [[nodiscard]] const T *begin() const noexcept { return cpu_data_.data(); } + + [[nodiscard]] const T *end() const noexcept { return cpu_data_.data() + cpu_data_.size(); } + + [[nodiscard]] const T &operator[](std::size_t idx) const { return cpu_data_[idx]; } + + // ReSharper disable once CppNonExplicitConversionOperator + operator std::span() const noexcept // NOLINT(*-explicit-constructor) + { + return { cpu_data_.data(), cpu_data_.size() }; + } + + protected: + // Serialization support: even if all of the code below runs on one + // locality only, we need to provide an (empty) implementation for the + // serialization as all arguments passed to actions have to support this. + friend class hpx::serialization::access; + + template + void serialize(Archive &ar, const unsigned int) + { + // clang-format off + ar & cpu_data_; + // clang-format on + } + + cpu_buffer_type cpu_data_; +}; + +/** + * A mutable version of const_tile_data. + * + * @tparam T Element type of the tile. See @ref const_tile_data + */ +template +class mutable_tile_data : public const_tile_data +{ + public: + using const_tile_data::const_tile_data; + + [[nodiscard]] T *data() const noexcept { return const_cast(this->cpu_data_.data()); } + + [[nodiscard]] T *begin() const noexcept { return const_cast(this->cpu_data_.data()); } + + [[nodiscard]] T *end() const noexcept { return const_cast(this->cpu_data_.data()) + this->cpu_data_.size(); } + + [[nodiscard]] T &operator[](std::size_t idx) const { return this->cpu_data_[idx]; } + + // ReSharper disable once CppNonExplicitConversionOperator + operator std::span() noexcept + { + return { this->cpu_data_.data(), this->cpu_data_.size() }; + } // NOLINT(*-explicit-constructor) +}; + +GPRAT_NS_END + +#endif diff --git a/core/src/cpu/adapter_cblas_fp32.cpp b/core/src/cpu/adapter_cblas_fp32.cpp index 2b7e5c12..acb3585c 100644 --- a/core/src/cpu/adapter_cblas_fp32.cpp +++ b/core/src/cpu/adapter_cblas_fp32.cpp @@ -13,26 +13,24 @@ GPRAT_NS_BEGIN // BLAS level 3 operations -vector_future potrf(vector_future f_A, const int N) +mutable_tile_data potrf(const mutable_tile_data &A, const int N) { - auto A = f_A.get(); // POTRF: in-place Cholesky decomposition of A // use spotrf2 recursive version for better stability LAPACKE_spotrf2(LAPACK_ROW_MAJOR, 'L', N, A.data(), N); // return factorized matrix L - return hpx::make_ready_future(A); + return A; } -vector_future trsm(vector_future f_L, - vector_future f_A, - const int N, - const int M, - const BLAS_TRANSPOSE transpose_L, - const BLAS_SIDE side_L) +mutable_tile_data +trsm(const const_tile_data &L, + const mutable_tile_data &A, + const int N, + const int M, + const BLAS_TRANSPOSE transpose_L, + const BLAS_SIDE side_L) { - auto L = f_L.get(); - auto A = f_A.get(); // TRSM constants const float alpha = 1.0; // TRSM: in-place solve L(^T) * X = A or X * L(^T) = A where L lower triangular @@ -49,36 +47,30 @@ vector_future trsm(vector_future f_L, N, A.data(), M); - // return vector - return hpx::make_ready_future(A); + return A; } -vector_future syrk(vector_future f_A, vector_future f_B, const int N) +mutable_tile_data syrk(const mutable_tile_data &A, const const_tile_data &B, const int N) { - auto B = f_B.get(); - auto A = f_A.get(); // SYRK constants const float alpha = -1.0; const float beta = 1.0; // SYRK:A = A - B * B^T cblas_ssyrk(CblasRowMajor, CblasLower, CblasNoTrans, N, N, alpha, B.data(), N, beta, A.data(), N); // return updated matrix A - return hpx::make_ready_future(A); + return A; } -vector_future -gemm(vector_future f_A, - vector_future f_B, - vector_future f_C, +mutable_tile_data +gemm(const const_tile_data &A, + const const_tile_data &B, + const mutable_tile_data &C, const int N, const int M, const int K, const BLAS_TRANSPOSE transpose_A, const BLAS_TRANSPOSE transpose_B) { - auto C = f_C.get(); - auto B = f_B.get(); - auto A = f_A.get(); // GEMM constants const float alpha = -1.0; const float beta = 1.0; @@ -99,15 +91,14 @@ gemm(vector_future f_A, C.data(), M); // return updated matrix C - return hpx::make_ready_future(C); + return C; } // BLAS level 2 operations -vector_future trsv(vector_future f_L, vector_future f_a, const int N, const BLAS_TRANSPOSE transpose_L) +mutable_tile_data +trsv(const const_tile_data &L, const mutable_tile_data &a, const int N, const BLAS_TRANSPOSE transpose_L) { - auto L = f_L.get(); - auto a = f_a.get(); // TRSV: In-place solve L(^T) * x = a where L lower triangular cblas_strsv(CblasRowMajor, CblasLower, @@ -119,20 +110,18 @@ vector_future trsv(vector_future f_L, vector_future f_a, const int N, const BLAS a.data(), 1); // return solution vector x - return hpx::make_ready_future(a); + return a; } -vector_future gemv(vector_future f_A, - vector_future f_a, - vector_future f_b, - const int N, - const int M, - const BLAS_ALPHA alpha, - const BLAS_TRANSPOSE transpose_A) +mutable_tile_data +gemv(const const_tile_data &A, + const const_tile_data &a, + const mutable_tile_data &b, + const int N, + const int M, + const BLAS_ALPHA alpha, + const BLAS_TRANSPOSE transpose_A) { - auto A = f_A.get(); - auto a = f_a.get(); - auto b = f_b.get(); // GEMV constants // const float alpha = -1.0; const float beta = 1.0; @@ -151,46 +140,50 @@ vector_future gemv(vector_future f_A, b.data(), 1); // return updated vector b - return hpx::make_ready_future(b); + return b; } -vector_future dot_diag_syrk(vector_future f_A, vector_future f_r, const int N, const int M) +mutable_tile_data +dot_diag_syrk(const const_tile_data &A, const mutable_tile_data &r, const int N, const int M) { - auto A = f_A.get(); - auto r = f_r.get(); + auto r_p = r.data(); + auto A_p = A.data(); // r = r + diag(A^T * A) for (std::size_t j = 0; j < static_cast(M); ++j) { // Extract the j-th column and compute the dot product with itself - r[j] += cblas_sdot(N, &A[j], M, &A[j], M); + r_p[j] += cblas_sdot(N, &A_p[j], M, &A_p[j], M); } - return hpx::make_ready_future(r); + return r; } -vector_future dot_diag_gemm(vector_future f_A, vector_future f_B, vector_future f_r, const int N, const int M) +mutable_tile_data +dot_diag_gemm(const const_tile_data &A, + const const_tile_data &B, + const mutable_tile_data &r, + const int N, + const int M) { - auto A = f_A.get(); - auto B = f_B.get(); - auto r = f_r.get(); + auto r_p = r.data(); + auto A_p = A.data(); + auto B_p = B.data(); // r = r + diag(A * B) for (std::size_t i = 0; i < static_cast(N); ++i) { - r[i] += cblas_sdot(M, &A[i * static_cast(M)], 1, &B[i], N); + r_p[i] += cblas_sdot(M, &A_p[i * static_cast(M)], 1, &B_p[i], N); } - return hpx::make_ready_future(r); + return r; } // BLAS level 1 operations -vector_future axpy(vector_future f_y, vector_future f_x, const int N) +mutable_tile_data axpy(const mutable_tile_data &y, const const_tile_data &x, const int N) { - auto y = f_y.get(); - auto x = f_x.get(); cblas_saxpy(N, -1.0, x.data(), 1, y.data(), 1); - return hpx::make_ready_future(y); + return y; } -float dot(std::vector a, std::vector b, const int N) +float dot(std::span a, std::span b, const int N) { // DOT: a * b return cblas_sdot(N, a.data(), 1, b.data(), 1); diff --git a/core/src/cpu/adapter_cblas_fp64.cpp b/core/src/cpu/adapter_cblas_fp64.cpp index 3cc15500..aeedb1c3 100644 --- a/core/src/cpu/adapter_cblas_fp64.cpp +++ b/core/src/cpu/adapter_cblas_fp64.cpp @@ -13,26 +13,24 @@ GPRAT_NS_BEGIN // BLAS level 3 operations -vector_future potrf(vector_future f_A, const int N) +mutable_tile_data potrf(const mutable_tile_data &A, const int N) { - auto A = f_A.get(); // POTRF: in-place Cholesky decomposition of A // use dpotrf2 recursive version for better stability LAPACKE_dpotrf2(LAPACK_ROW_MAJOR, 'L', N, A.data(), N); // return factorized matrix L - return hpx::make_ready_future(A); + return A; } -vector_future trsm(vector_future f_L, - vector_future f_A, - const int N, - const int M, - const BLAS_TRANSPOSE transpose_L, - const BLAS_SIDE side_L) +mutable_tile_data +trsm(const const_tile_data &L, + const mutable_tile_data &A, + const int N, + const int M, + const BLAS_TRANSPOSE transpose_L, + const BLAS_SIDE side_L) { - auto L = f_L.get(); - auto A = f_A.get(); // TRSM constants const double alpha = 1.0; // TRSM: in-place solve L(^T) * X = A or X * L(^T) = A where L lower triangular @@ -50,35 +48,30 @@ vector_future trsm(vector_future f_L, A.data(), M); // return vector - return hpx::make_ready_future(A); + return A; } -vector_future syrk(vector_future f_A, vector_future f_B, const int N) +mutable_tile_data syrk(const mutable_tile_data &A, const const_tile_data &B, const int N) { - auto B = f_B.get(); - auto A = f_A.get(); // SYRK constants const double alpha = -1.0; const double beta = 1.0; // SYRK:A = A - B * B^T cblas_dsyrk(CblasRowMajor, CblasLower, CblasNoTrans, N, N, alpha, B.data(), N, beta, A.data(), N); // return updated matrix A - return hpx::make_ready_future(A); + return A; } -vector_future -gemm(vector_future f_A, - vector_future f_B, - vector_future f_C, +mutable_tile_data +gemm(const const_tile_data &A, + const const_tile_data &B, + const mutable_tile_data &C, const int N, const int M, const int K, const BLAS_TRANSPOSE transpose_A, const BLAS_TRANSPOSE transpose_B) { - auto C = f_C.get(); - auto B = f_B.get(); - auto A = f_A.get(); // GEMM constants const double alpha = -1.0; const double beta = 1.0; @@ -99,15 +92,14 @@ gemm(vector_future f_A, C.data(), M); // return updated matrix C - return hpx::make_ready_future(C); + return C; } // BLAS level 2 operations -vector_future trsv(vector_future f_L, vector_future f_a, const int N, const BLAS_TRANSPOSE transpose_L) +mutable_tile_data trsv( + const const_tile_data &L, const mutable_tile_data &a, const int N, const BLAS_TRANSPOSE transpose_L) { - auto L = f_L.get(); - auto a = f_a.get(); // TRSV: In-place solve L(^T) * x = a where L lower triangular cblas_dtrsv(CblasRowMajor, CblasLower, @@ -119,20 +111,18 @@ vector_future trsv(vector_future f_L, vector_future f_a, const int N, const BLAS a.data(), 1); // return solution vector x - return hpx::make_ready_future(a); + return a; } -vector_future gemv(vector_future f_A, - vector_future f_a, - vector_future f_b, - const int N, - const int M, - const BLAS_ALPHA alpha, - const BLAS_TRANSPOSE transpose_A) +mutable_tile_data +gemv(const const_tile_data &A, + const const_tile_data &a, + const mutable_tile_data &b, + const int N, + const int M, + const BLAS_ALPHA alpha, + const BLAS_TRANSPOSE transpose_A) { - auto A = f_A.get(); - auto a = f_a.get(); - auto b = f_b.get(); // GEMV constants // const double alpha = -1.0; const double beta = 1.0; @@ -151,46 +141,50 @@ vector_future gemv(vector_future f_A, b.data(), 1); // return updated vector b - return hpx::make_ready_future(b); + return b; } -vector_future dot_diag_syrk(vector_future f_A, vector_future f_r, const int N, const int M) +mutable_tile_data +dot_diag_syrk(const const_tile_data &A, const mutable_tile_data &r, const int N, const int M) { - auto A = f_A.get(); - auto r = f_r.get(); + auto r_p = r.data(); + auto A_p = A.data(); // r = r + diag(A^T * A) for (std::size_t j = 0; j < static_cast(M); ++j) { // Extract the j-th column and compute the dot product with itself - r[j] += cblas_ddot(N, &A[j], M, &A[j], M); + r_p[j] += cblas_ddot(N, &A_p[j], M, &A_p[j], M); } - return hpx::make_ready_future(r); + return r; } -vector_future dot_diag_gemm(vector_future f_A, vector_future f_B, vector_future f_r, const int N, const int M) +mutable_tile_data +dot_diag_gemm(const const_tile_data &A, + const const_tile_data &B, + const mutable_tile_data &r, + const int N, + const int M) { - auto A = f_A.get(); - auto B = f_B.get(); - auto r = f_r.get(); + auto r_p = r.data(); + auto A_p = A.data(); + auto B_p = B.data(); // r = r + diag(A * B) for (std::size_t i = 0; i < static_cast(N); ++i) { - r[i] += cblas_ddot(M, &A[i * static_cast(M)], 1, &B[i], N); + r_p[i] += cblas_ddot(M, &A_p[i * static_cast(M)], 1, &B_p[i], N); } - return hpx::make_ready_future(r); + return r; } // BLAS level 1 operations -vector_future axpy(vector_future f_y, vector_future f_x, const int N) +mutable_tile_data axpy(const mutable_tile_data &y, const const_tile_data &x, const int N) { - auto y = f_y.get(); - auto x = f_x.get(); cblas_daxpy(N, -1.0, x.data(), 1, y.data(), 1); - return hpx::make_ready_future(y); + return y; } -double dot(std::vector a, std::vector b, const int N) +double dot(std::span a, std::span b, const int N) { // DOT: a * b return cblas_ddot(N, a.data(), 1, b.data(), 1); diff --git a/core/src/cpu/gp_algorithms.cpp b/core/src/cpu/gp_algorithms.cpp index 8b42e12a..b02dfe4e 100644 --- a/core/src/cpu/gp_algorithms.cpp +++ b/core/src/cpu/gp_algorithms.cpp @@ -1,7 +1,8 @@ #include "gprat/cpu/gp_algorithms.hpp" +#include "gprat/tile_data.hpp" + #include -#include GPRAT_NS_BEGIN @@ -10,176 +11,162 @@ namespace cpu // Tile generation -double compute_covariance_function(std::size_t i_global, - std::size_t j_global, - std::size_t n_regressors, +double compute_covariance_function(std::size_t n_regressors, const SEKParams &sek_params, - const std::vector &i_input, - const std::vector &j_input) + std::span i_input, + std::span j_input) { // k(z_i,z_j) = vertical_lengthscale * exp(-0.5 / lengthscale^2 * (z_i - z_j)^2) double distance = 0.0; - double z_ik_minus_z_jk; - for (std::size_t k = 0; k < n_regressors; k++) { - z_ik_minus_z_jk = i_input[i_global + k] - j_input[j_global + k]; + const double z_ik_minus_z_jk = i_input[k] - j_input[k]; distance += z_ik_minus_z_jk * z_ik_minus_z_jk; } + return sek_params.vertical_lengthscale * exp(-0.5 / (sek_params.lengthscale * sek_params.lengthscale) * distance); } -std::vector gen_tile_covariance( +mutable_tile_data gen_tile_covariance( std::size_t row, std::size_t col, std::size_t N, std::size_t n_regressors, const SEKParams &sek_params, - const std::vector &input) + std::span input) { - std::size_t i_global, j_global; - double covariance_function; - // Preallocate required memory - std::vector tile; - tile.reserve(N * N); - // Compute entries + mutable_tile_data tile(N * N); for (std::size_t i = 0; i < N; i++) { - i_global = N * row + i; + const std::size_t i_global = N * row + i; for (std::size_t j = 0; j < N; j++) { - j_global = N * col + j; + const std::size_t j_global = N * col + j; + // compute covariance function - covariance_function = - compute_covariance_function(i_global, j_global, n_regressors, sek_params, input, input); + auto covariance_function = compute_covariance_function( + n_regressors, sek_params, input.subspan(i_global, n_regressors), input.subspan(j_global, n_regressors)); if (i_global == j_global) { // noise variance on diagonal covariance_function += sek_params.noise_variance; } - tile.push_back(covariance_function); + + tile.data()[i * N + j] = covariance_function; } } return tile; } -std::vector gen_tile_full_prior_covariance( +mutable_tile_data gen_tile_full_prior_covariance( std::size_t row, std::size_t col, std::size_t N, std::size_t n_regressors, const SEKParams &sek_params, - const std::vector &input) + std::span input) { - std::size_t i_global, j_global; - // Preallocate required memory - std::vector tile; - tile.reserve(N * N); - // Compute entries + mutable_tile_data tile(N * N); for (std::size_t i = 0; i < N; i++) { - i_global = N * row + i; + const std::size_t i_global = N * row + i; for (std::size_t j = 0; j < N; j++) { - j_global = N * col + j; + const std::size_t j_global = N * col + j; // compute covariance function - tile.push_back(compute_covariance_function(i_global, j_global, n_regressors, sek_params, input, input)); + tile.data()[i * N + j] = compute_covariance_function( + n_regressors, sek_params, input.subspan(i_global, n_regressors), input.subspan(j_global, n_regressors)); } } return tile; } -std::vector gen_tile_prior_covariance( +mutable_tile_data gen_tile_prior_covariance( std::size_t row, std::size_t col, std::size_t N, std::size_t n_regressors, const SEKParams &sek_params, - const std::vector &input) + std::span input) { - std::size_t i_global, j_global; - // Preallocate required memory - std::vector tile; - tile.reserve(N); - // Compute entries + mutable_tile_data tile(N); for (std::size_t i = 0; i < N; i++) { - i_global = N * row + i; - j_global = N * col + i; + const std::size_t i_global = N * row + i; + const std::size_t j_global = N * col + i; // compute covariance function - tile.push_back(compute_covariance_function(i_global, j_global, n_regressors, sek_params, input, input)); + tile.data()[i] = compute_covariance_function( + n_regressors, sek_params, input.subspan(i_global, n_regressors), input.subspan(j_global, n_regressors)); } return tile; } -std::vector gen_tile_cross_covariance( +mutable_tile_data gen_tile_cross_covariance( std::size_t row, std::size_t col, std::size_t N_row, std::size_t N_col, std::size_t n_regressors, const SEKParams &sek_params, - const std::vector &row_input, - const std::vector &col_input) + std::span row_input, + std::span col_input) { - std::size_t i_global, j_global; - // Preallocate required memory - std::vector tile; - tile.reserve(N_row * N_col); - // Compute entries + mutable_tile_data tile(N_row * N_col); for (std::size_t i = 0; i < N_row; i++) { - i_global = N_row * row + i; + std::size_t i_global = N_row * row + i; for (std::size_t j = 0; j < N_col; j++) { - j_global = N_col * col + j; + std::size_t j_global = N_col * col + j; // compute covariance function - tile.push_back( - compute_covariance_function(i_global, j_global, n_regressors, sek_params, row_input, col_input)); + tile.data()[i * N_col + j] = compute_covariance_function( + n_regressors, + sek_params, + row_input.subspan(i_global, n_regressors), + col_input.subspan(j_global, n_regressors)); } } return tile; } -std::vector gen_tile_transpose(std::size_t N_row, std::size_t N_col, const std::vector &tile) +mutable_tile_data gen_tile_transpose(std::size_t N_row, std::size_t N_col, std::span tile) { - // Preallocate required memory - std::vector transposed; - transposed.reserve(N_row * N_col); + mutable_tile_data transposed(N_row * N_col); // Transpose entries for (std::size_t j = 0; j < N_col; j++) { for (std::size_t i = 0; i < N_row; ++i) { // Mapping (i, j) in the original tile to (j, i) in the transposed tile - transposed.push_back(tile[i * N_col + j]); + transposed.data()[j * N_row + i] = tile[i * N_col + j]; } } return transposed; } -std::vector gen_tile_output(std::size_t row, std::size_t N, const std::vector &output) +mutable_tile_data gen_tile_output(std::size_t row, std::size_t N, std::span output) { - // Preallocate required memory - std::vector tile; - tile.reserve(N); - // Copy entries - std::copy(output.begin() + static_cast(N * row), - output.begin() + static_cast(N * (row + 1)), - std::back_inserter(tile)); + mutable_tile_data tile(N); + std::copy(output.begin() + (N * row), output.begin() + (N * (row + 1)), tile.data()); return tile; } -std::vector gen_tile_zeros(std::size_t N) { return std::vector(N, 0.0); } +mutable_tile_data gen_tile_zeros(std::size_t N) +{ + mutable_tile_data tile(N); + std::fill_n(tile.data(), N, 0.0); + return tile; +} -std::vector gen_tile_identity(std::size_t N) +mutable_tile_data gen_tile_identity(std::size_t N) { + mutable_tile_data tile(N * N); // Initialize zero tile - std::vector tile(N * N, 0.0); + std::fill_n(tile.data(), N * N, 0.0); // Fill diagonal with ones for (std::size_t i = 0; i < N; i++) { - tile[i * N + i] = 1.0; + tile.data()[i * N + i] = 1.0; } return tile; } diff --git a/core/src/cpu/gp_functions.cpp b/core/src/cpu/gp_functions.cpp index 9f16f319..4b70c691 100644 --- a/core/src/cpu/gp_functions.cpp +++ b/core/src/cpu/gp_functions.cpp @@ -3,27 +3,25 @@ #include "gprat/cpu/gp_algorithms.hpp" #include "gprat/cpu/gp_optimizer.hpp" #include "gprat/cpu/tiled_algorithms.hpp" +#include "gprat/detail/async_helpers.hpp" #include GPRAT_NS_BEGIN -using Tiled_matrix = std::vector>>; -using Tiled_vector = std::vector>>; - namespace cpu { /////////////////////////////////////////////////////////////////////////// // PREDICT -std::vector> +std::vector> cholesky(const std::vector &training_input, const SEKParams &sek_params, int n_tiles, int n_tile_size, int n_regressors) { - std::vector> result; + std::vector> result; // Tiled future data structures Tiled_matrix K_tiles; // Tiled covariance matrix @@ -37,14 +35,8 @@ cholesky(const std::vector &training_input, { for (std::size_t j = 0; j <= i; j++) { - K_tiles[i * static_cast(n_tiles) + j] = hpx::async( - hpx::annotated_function(gen_tile_covariance, "assemble_tiled_K"), - i, - j, - n_tile_size, - n_regressors, - sek_params, - training_input); + K_tiles[i * static_cast(n_tiles) + j] = detail::named_async( + "assemble_tiled_K", i, j, n_tile_size, n_regressors, sek_params, training_input); } } @@ -113,43 +105,29 @@ predict(const std::vector &training_input, { for (std::size_t j = 0; j <= i; j++) { - K_tiles[i * static_cast(n_tiles) + j] = hpx::async( - hpx::annotated_function(gen_tile_covariance, "assemble_tiled_K"), - i, - j, - n_tile_size, - n_regressors, - sek_params, - training_input); + K_tiles[i * static_cast(n_tiles) + j] = detail::named_async( + "assemble_tiled_K", i, j, n_tile_size, n_regressors, sek_params, training_input); } } for (std::size_t i = 0; i < static_cast(n_tiles); i++) { - alpha_tiles.push_back(hpx::async( - hpx::annotated_function(gen_tile_output, "assemble_tiled_alpha"), i, n_tile_size, training_output)); + alpha_tiles.push_back( + detail::named_async("assemble_tiled_alpha", i, n_tile_size, training_output)); } for (std::size_t i = 0; i < static_cast(m_tiles); i++) { for (std::size_t j = 0; j < static_cast(n_tiles); j++) { - cross_covariance_tiles.push_back(hpx::async( - hpx::annotated_function(gen_tile_cross_covariance, "assemble_pred"), - i, - j, - m_tile_size, - n_tile_size, - n_regressors, - sek_params, - test_input, - training_input)); + cross_covariance_tiles.push_back(detail::named_async( + "assemble_pred", i, j, m_tile_size, n_tile_size, n_regressors, sek_params, test_input, training_input)); } } for (std::size_t i = 0; i < static_cast(m_tiles); i++) { - prediction_tiles.push_back(hpx::async(hpx::annotated_function(gen_tile_zeros, "assemble_tiled"), m_tile_size)); + prediction_tiles.push_back(detail::named_async("assemble_tiled", m_tile_size)); } /////////////////////////////////////////////////////////////////////////// @@ -177,7 +155,7 @@ predict(const std::vector &training_input, for (std::size_t i = 0; i < static_cast(m_tiles); i++) { auto tile = prediction_tiles[i].get(); - std::copy(tile.begin(), tile.end(), std::back_inserter(prediction_result)); + std::copy_n(tile.data(), tile.size(), std::back_inserter(prediction_result)); } return prediction_result; } @@ -247,14 +225,8 @@ std::vector> predict_with_uncertainty( { for (std::size_t j = 0; j <= i; j++) { - K_tiles[i * static_cast(n_tiles) + j] = hpx::async( - hpx::annotated_function(gen_tile_covariance, "assemble_tiled_K"), - i, - j, - n_tile_size, - n_regressors, - sek_params, - training_input); + K_tiles[i * static_cast(n_tiles) + j] = detail::named_async( + "assemble_tiled_K", i, j, n_tile_size, n_regressors, sek_params, training_input); } } @@ -366,14 +338,14 @@ std::vector> predict_with_uncertainty( for (std::size_t i = 0; i < static_cast(m_tiles); i++) { auto tile = prediction_tiles[i].get(); - std::copy(tile.begin(), tile.end(), std::back_inserter(prediction_result)); + std::copy_n(tile.begin(), tile.size(), std::back_inserter(prediction_result)); } // Synchronize uncertainty for (std::size_t i = 0; i < static_cast(m_tiles); i++) { auto tile = uncertainty_tiles[i].get(); - std::copy(tile.begin(), tile.end(), std::back_inserter(uncertainty_result)); + std::copy_n(tile.begin(), tile.size(), std::back_inserter(uncertainty_result)); } return std::vector>{ std::move(prediction_result), std::move(uncertainty_result) }; @@ -676,9 +648,9 @@ double compute_loss(const std::vector &training_input, std::vector optimize(const std::vector &training_input, const std::vector &training_output, - int n_tiles, - int n_tile_size, - int n_regressors, + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors, const AdamParams &adam_params, SEKParams &sek_params, std::vector trainable_params) @@ -732,18 +704,18 @@ optimize(const std::vector &training_input, // Preallocate memory losses.reserve(static_cast(adam_params.opt_iter)); - y_tiles.reserve(static_cast(n_tiles)); + y_tiles.reserve(n_tiles); - alpha_tiles.resize(static_cast(n_tiles)); // for now resize since reset in loop - K_inv_tiles.resize(static_cast(n_tiles * n_tiles)); // for now resize since reset in loop + alpha_tiles.resize(n_tiles); // for now resize since reset in loop + K_inv_tiles.resize(n_tiles * n_tiles); // for now resize since reset in loop - K_tiles.resize(static_cast(n_tiles * n_tiles)); // No reserve because of triangular structure - grad_v_tiles.resize(static_cast(n_tiles * n_tiles)); // No reserve because of triangular structure - grad_l_tiles.resize(static_cast(n_tiles * n_tiles)); // No reserve because of triangular structure + K_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure + grad_v_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure + grad_l_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure /////////////////////////////////////////////////////////////////////////// // Launch asynchronous assembly of output y - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { y_tiles.push_back( hpx::async(hpx::annotated_function(gen_tile_output, "assemble_y"), i, n_tile_size, training_output)); @@ -757,150 +729,92 @@ optimize(const std::vector &training_input, // Launch asynchronous assembly of tiled covariance matrix, derivative of covariance matrix // vector w.r.t. to vertical lengthscale and derivative of covariance // matrix vector w.r.t. to lengthscale - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { for (std::size_t j = 0; j <= i; j++) { // Compute the distance (z_i - z_j) of K entries to reuse - hpx::shared_future> cov_dists = hpx::async( - hpx::annotated_function(gen_tile_distance, "assemble_cov_dist"), - i, - j, - n_tile_size, - n_regressors, - sek_params, - training_input); + hpx::shared_future> cov_dists = detail::named_async( + "assemble_cov_dist", i, j, n_tile_size, n_regressors, sek_params, training_input); - K_tiles[i * static_cast(n_tiles) + j] = hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&gen_tile_covariance_with_distance), "assemble_K"), - i, - j, - n_tile_size, - sek_params, - cov_dists); + K_tiles[i * n_tiles + j] = detail::named_dataflow( + "assemble_K", i, j, n_tile_size, sek_params, cov_dists); if (trainable_params[0]) { - grad_l_tiles[i * static_cast(n_tiles) + j] = hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&gen_tile_grad_l), "assemble_gradl"), - n_tile_size, - sek_params, - cov_dists); + grad_l_tiles[i * n_tiles + j] = + detail::named_dataflow("assemble_gradl", n_tile_size, sek_params, cov_dists); if (i != j) { - grad_l_tiles[j * static_cast(n_tiles) + i] = hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&gen_tile_transpose), "assemble_gradl_t"), - n_tile_size, - n_tile_size, - grad_l_tiles[i * static_cast(n_tiles) + j]); + grad_l_tiles[j * n_tiles + i] = detail::named_dataflow( + "assemble_gradl_t", n_tile_size, n_tile_size, grad_l_tiles[i * n_tiles + j]); } } if (trainable_params[1]) { - grad_v_tiles[i * static_cast(n_tiles) + j] = hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&gen_tile_grad_v), "assemble_gradv"), - n_tile_size, - sek_params, - cov_dists); + grad_v_tiles[i * n_tiles + j] = + detail::named_dataflow("assemble_gradv", n_tile_size, sek_params, cov_dists); if (i != j) { - grad_v_tiles[j * static_cast(n_tiles) + i] = hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&gen_tile_transpose), "assemble_gradv_t"), - n_tile_size, - n_tile_size, - grad_v_tiles[i * static_cast(n_tiles) + j]); + grad_v_tiles[j * n_tiles + i] = detail::named_dataflow( + "assemble_gradv_t", n_tile_size, n_tile_size, grad_v_tiles[i * n_tiles + j]); } } } } // Assembly with reallocation -> optimize to only set existing values - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { - alpha_tiles[i] = hpx::async(hpx::annotated_function(gen_tile_zeros, "assemble_tiled"), n_tile_size); + alpha_tiles[i] = detail::named_async("assemble_tiled", n_tile_size); } - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { - for (std::size_t j = 0; j < static_cast(n_tiles); j++) + for (std::size_t j = 0; j < n_tiles; j++) { if (i == j) { - K_inv_tiles[i * static_cast(n_tiles) + j] = - hpx::async(hpx::annotated_function(gen_tile_identity, "assemble_identity_matrix"), n_tile_size); + K_inv_tiles[i * n_tiles + j] = + detail::named_async("assemble_identity_matrix", n_tile_size); } else { - K_inv_tiles[i * static_cast(n_tiles) + j] = hpx::async( - hpx::annotated_function(gen_tile_zeros, "assemble_identity_matrix"), n_tile_size * n_tile_size); + K_inv_tiles[i * n_tiles + j] = + detail::named_async("assemble_identity_matrix", n_tile_size * n_tile_size); } } } /////////////////////////////////////////////////////////////////////////// // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(K_tiles, n_tile_size, static_cast(n_tiles)); + right_looking_cholesky_tiled(K_tiles, n_tile_size, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous compute K^-1 through L* (L^T * X) = I - forward_solve_tiled_matrix( - K_tiles, - K_inv_tiles, - n_tile_size, - n_tile_size, - static_cast(n_tiles), - static_cast(n_tiles)); - backward_solve_tiled_matrix( - K_tiles, - K_inv_tiles, - n_tile_size, - n_tile_size, - static_cast(n_tiles), - static_cast(n_tiles)); + forward_solve_tiled_matrix(K_tiles, K_inv_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); + backward_solve_tiled_matrix(K_tiles, K_inv_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous compute beta = inv(K) * y - matrix_vector_tiled( - K_inv_tiles, - y_tiles, - alpha_tiles, - n_tile_size, - n_tile_size, - static_cast(n_tiles), - static_cast(n_tiles)); + matrix_vector_tiled(K_inv_tiles, y_tiles, alpha_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous loss computation where // loss(theta) = 0.5 * ( log(det(K)) - y^T * K^-1 * y - N * log(2 * pi) ) - compute_loss_tiled(K_tiles, alpha_tiles, y_tiles, loss_value, n_tile_size, static_cast(n_tiles)); + compute_loss_tiled(K_tiles, alpha_tiles, y_tiles, loss_value, n_tile_size, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous update of the hyperparameters if (trainable_params[0]) { // lengthscale update_hyperparameter_tiled( - K_inv_tiles, - grad_l_tiles, - alpha_tiles, - adam_params, - sek_params, - n_tile_size, - static_cast(n_tiles), - iter, - 0); + K_inv_tiles, grad_l_tiles, alpha_tiles, adam_params, sek_params, n_tile_size, n_tiles, iter, 0); } if (trainable_params[1]) { // vertical_lengthscale update_hyperparameter_tiled( - K_inv_tiles, - grad_v_tiles, - alpha_tiles, - adam_params, - sek_params, - n_tile_size, - static_cast(n_tiles), - iter, - 1); + K_inv_tiles, grad_v_tiles, alpha_tiles, adam_params, sek_params, n_tile_size, n_tiles, iter, 1); } if (trainable_params[2]) { // noise_variance @@ -911,7 +825,7 @@ optimize(const std::vector &training_input, adam_params, sek_params, n_tile_size, - static_cast(n_tiles), + n_tiles, iter, 2); } @@ -924,13 +838,13 @@ optimize(const std::vector &training_input, double optimize_step(const std::vector &training_input, const std::vector &training_output, - int n_tiles, - int n_tile_size, - int n_regressors, + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors, AdamParams &adam_params, SEKParams &sek_params, std::vector trainable_params, - int iter) + std::size_t iter) { /* * - Hyperparameters theta={v, l, v_n} @@ -976,18 +890,18 @@ double optimize_step(const std::vector &training_input, Tiled_matrix grad_l_tiles; // Tiled covariance with gradient l // Preallocate memory - y_tiles.reserve(static_cast(n_tiles)); + y_tiles.reserve(n_tiles); - alpha_tiles.resize(static_cast(n_tiles)); // for now resize since reset in loop - K_inv_tiles.resize(static_cast(n_tiles * n_tiles)); // for now resize since reset in loop + alpha_tiles.resize(n_tiles); // for now resize since reset in loop + K_inv_tiles.resize(n_tiles * n_tiles); // for now resize since reset in loop - K_tiles.resize(static_cast(n_tiles * n_tiles)); // No reserve because of triangular structure - grad_v_tiles.resize(static_cast(n_tiles * n_tiles)); // No reserve because of triangular structure - grad_l_tiles.resize(static_cast(n_tiles * n_tiles)); // No reserve because of triangular structure + K_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure + grad_v_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure + grad_l_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure /////////////////////////////////////////////////////////////////////////// // Launch asynchronous assembly of output y - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { y_tiles.push_back( hpx::async(hpx::annotated_function(gen_tile_output, "assemble_y"), i, n_tile_size, training_output)); @@ -999,12 +913,12 @@ double optimize_step(const std::vector &training_input, // Launch asynchronous assembly of tiled covariance matrix, derivative of covariance matrix // vector w.r.t. to vertical lengthscale and derivative of covariance // matrix vector w.r.t. to lengthscale - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { for (std::size_t j = 0; j <= i; j++) { // Compute the distance (z_i - z_j) of K entries to reuse - hpx::shared_future> cov_dists = hpx::async( + hpx::shared_future> cov_dists = hpx::async( hpx::annotated_function(gen_tile_distance, "assemble_cov_dist"), i, j, @@ -1013,7 +927,7 @@ double optimize_step(const std::vector &training_input, sek_params, training_input); - K_tiles[i * static_cast(n_tiles) + j] = hpx::dataflow( + K_tiles[i * n_tiles + j] = hpx::dataflow( hpx::annotated_function(hpx::unwrapping(&gen_tile_covariance_with_distance), "assemble_K"), i, j, @@ -1023,58 +937,58 @@ double optimize_step(const std::vector &training_input, if (trainable_params[0]) { - grad_l_tiles[i * static_cast(n_tiles) + j] = hpx::dataflow( + grad_l_tiles[i * n_tiles + j] = hpx::dataflow( hpx::annotated_function(hpx::unwrapping(&gen_tile_grad_l), "assemble_gradl"), n_tile_size, sek_params, cov_dists); if (i != j) { - grad_l_tiles[j * static_cast(n_tiles) + i] = hpx::dataflow( + grad_l_tiles[j * n_tiles + i] = hpx::dataflow( hpx::annotated_function(hpx::unwrapping(&gen_tile_transpose), "assemble_gradl_t"), n_tile_size, n_tile_size, - grad_l_tiles[i * static_cast(n_tiles) + j]); + grad_l_tiles[i * n_tiles + j]); } } if (trainable_params[1]) { - grad_v_tiles[i * static_cast(n_tiles) + j] = hpx::dataflow( + grad_v_tiles[i * n_tiles + j] = hpx::dataflow( hpx::annotated_function(hpx::unwrapping(&gen_tile_grad_v), "assemble_gradv"), n_tile_size, sek_params, cov_dists); if (i != j) { - grad_v_tiles[j * static_cast(n_tiles) + i] = hpx::dataflow( + grad_v_tiles[j * n_tiles + i] = hpx::dataflow( hpx::annotated_function(hpx::unwrapping(&gen_tile_transpose), "assemble_gradv_t"), n_tile_size, n_tile_size, - grad_v_tiles[i * static_cast(n_tiles) + j]); + grad_v_tiles[i * n_tiles + j]); } } } } // Assembly with reallocation -> optimize to only set existing values - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { alpha_tiles[i] = hpx::async(hpx::annotated_function(gen_tile_zeros, "assemble_tiled"), n_tile_size); } - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { - for (std::size_t j = 0; j < static_cast(n_tiles); j++) + for (std::size_t j = 0; j < n_tiles; j++) { if (i == j) { - K_inv_tiles[i * static_cast(n_tiles) + j] = + K_inv_tiles[i * n_tiles + j] = hpx::async(hpx::annotated_function(gen_tile_identity, "assemble_identity_matrix"), n_tile_size); } else { - K_inv_tiles[i * static_cast(n_tiles) + j] = hpx::async( + K_inv_tiles[i * n_tiles + j] = hpx::async( hpx::annotated_function(gen_tile_zeros, "assemble_identity_matrix"), n_tile_size * n_tile_size); } } @@ -1082,68 +996,33 @@ double optimize_step(const std::vector &training_input, /////////////////////////////////////////////////////////////////////////// // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(K_tiles, n_tile_size, static_cast(n_tiles)); + right_looking_cholesky_tiled(K_tiles, n_tile_size, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous compute K^-1 through L* (L^T * X) = I - forward_solve_tiled_matrix( - K_tiles, - K_inv_tiles, - n_tile_size, - n_tile_size, - static_cast(n_tiles), - static_cast(n_tiles)); - backward_solve_tiled_matrix( - K_tiles, - K_inv_tiles, - n_tile_size, - n_tile_size, - static_cast(n_tiles), - static_cast(n_tiles)); + forward_solve_tiled_matrix(K_tiles, K_inv_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); + backward_solve_tiled_matrix(K_tiles, K_inv_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous compute beta = inv(K) * y - matrix_vector_tiled( - K_inv_tiles, - y_tiles, - alpha_tiles, - n_tile_size, - n_tile_size, - static_cast(n_tiles), - static_cast(n_tiles)); + matrix_vector_tiled(K_inv_tiles, y_tiles, alpha_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous loss computation where // loss(theta) = 0.5 * ( log(det(K)) - y^T * K^-1 * y - N * log(2 * pi) ) - compute_loss_tiled(K_tiles, alpha_tiles, y_tiles, loss_value, n_tile_size, static_cast(n_tiles)); + compute_loss_tiled(K_tiles, alpha_tiles, y_tiles, loss_value, n_tile_size, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous update of the hyperparameters if (trainable_params[0]) { // lengthscale update_hyperparameter_tiled( - K_inv_tiles, - grad_l_tiles, - alpha_tiles, - adam_params, - sek_params, - n_tile_size, - static_cast(n_tiles), - static_cast(iter), - 0); + K_inv_tiles, grad_l_tiles, alpha_tiles, adam_params, sek_params, n_tile_size, n_tiles, iter, 0); } if (trainable_params[1]) { // vertical_lengthscale update_hyperparameter_tiled( - K_inv_tiles, - grad_v_tiles, - alpha_tiles, - adam_params, - sek_params, - n_tile_size, - static_cast(n_tiles), - static_cast(iter), - 1); + K_inv_tiles, grad_v_tiles, alpha_tiles, adam_params, sek_params, n_tile_size, n_tiles, iter, 1); } if (trainable_params[2]) { // noise_variance @@ -1154,8 +1033,8 @@ double optimize_step(const std::vector &training_input, adam_params, sek_params, n_tile_size, - static_cast(n_tiles), - static_cast(iter), + n_tiles, + iter, 2); } return loss_value.get(); diff --git a/core/src/cpu/gp_optimizer.cpp b/core/src/cpu/gp_optimizer.cpp index 081c037e..7c1c76f7 100644 --- a/core/src/cpu/gp_optimizer.cpp +++ b/core/src/cpu/gp_optimizer.cpp @@ -2,6 +2,7 @@ #include "gprat/cpu/adapter_cblas_fp64.hpp" +#include #include #include @@ -49,17 +50,15 @@ double compute_covariance_distance(std::size_t i_global, { // -0.5*lengthscale^2*(z_i-z_j)^2 double distance = 0.0; - double z_ik_minus_z_jk; - for (std::size_t k = 0; k < n_regressors; k++) { - z_ik_minus_z_jk = i_input[i_global + k] - j_input[j_global + k]; + const double z_ik_minus_z_jk = i_input[i_global + k] - j_input[j_global + k]; distance += z_ik_minus_z_jk * z_ik_minus_z_jk; } return -0.5 / (sek_params.lengthscale * sek_params.lengthscale) * distance; } -std::vector gen_tile_distance( +mutable_tile_data gen_tile_distance( std::size_t row, std::size_t col, std::size_t N, @@ -67,80 +66,81 @@ std::vector gen_tile_distance( const SEKParams &sek_params, const std::vector &input) { - std::size_t i_global, j_global; // Preallocate memory - std::vector tile; - tile.reserve(N * N); + mutable_tile_data tile(N * N); for (std::size_t i = 0; i < N; i++) { - i_global = N * row + i; + const std::size_t i_global = N * row + i; for (std::size_t j = 0; j < N; j++) { - j_global = N * col + j; + const std::size_t j_global = N * col + j; // compute covariance function - tile.push_back(compute_covariance_distance(i_global, j_global, n_regressors, sek_params, input, input)); + tile.data()[i * N + j] = + compute_covariance_distance(i_global, j_global, n_regressors, sek_params, input, input); } } return tile; } -std::vector gen_tile_covariance_with_distance( - std::size_t row, std::size_t col, std::size_t N, const SEKParams &sek_params, const std::vector &distance) +mutable_tile_data gen_tile_covariance_with_distance( + std::size_t row, + std::size_t col, + std::size_t N, + const SEKParams &sek_params, + const const_tile_data &distance) { - std::size_t i_global, j_global; - double covariance; // Preallocate required memory - std::vector tile; - tile.reserve(N * N); + mutable_tile_data tile(N * N); for (std::size_t i = 0; i < N; i++) { - i_global = N * row + i; + const std::size_t i_global = N * row + i; for (std::size_t j = 0; j < N; j++) { - j_global = N * col + j; + const std::size_t j_global = N * col + j; // compute covariance function - covariance = sek_params.vertical_lengthscale * exp(distance[i * N + j]); + double covariance = sek_params.vertical_lengthscale * exp(distance.data()[i * N + j]); if (i_global == j_global) { // noise variance on diagonal covariance += sek_params.noise_variance; } - tile.push_back(covariance); + tile.data()[i * N + j] = covariance; } } return tile; } -std::vector gen_tile_grad_v(std::size_t N, const SEKParams &sek_params, const std::vector &distance) +mutable_tile_data +gen_tile_grad_v(std::size_t N, const SEKParams &sek_params, const const_tile_data &distance) { // Preallocate required memory - std::vector tile; - tile.reserve(N * N); + mutable_tile_data tile(N * N); double hyperparam_der = compute_sigmoid(to_unconstrained(sek_params.vertical_lengthscale, false)); for (std::size_t i = 0; i < N; i++) { for (std::size_t j = 0; j < N; j++) { // compute derivative - tile.push_back(exp(distance[i * N + j]) * hyperparam_der); + tile.data()[i * N + j] = exp(distance.data()[i * N + j]) * hyperparam_der; } } return tile; } -std::vector gen_tile_grad_l(std::size_t N, const SEKParams &sek_params, const std::vector &distance) +mutable_tile_data +gen_tile_grad_l(std::size_t N, const SEKParams &sek_params, const const_tile_data &distance) { // Preallocate required memory - std::vector tile; - tile.reserve(N * N); - double hyperparam_der = compute_sigmoid(to_unconstrained(sek_params.lengthscale, false)); - double factor = -2.0 * sek_params.vertical_lengthscale / sek_params.lengthscale; + mutable_tile_data tile(N * N); + const double hyperparam_der = compute_sigmoid(to_unconstrained(sek_params.lengthscale, false)); + const double factor = -2.0 * sek_params.vertical_lengthscale / sek_params.lengthscale; for (std::size_t i = 0; i < N; i++) { for (std::size_t j = 0; j < N; j++) { // compute derivative - tile.push_back(factor * distance[i * N + j] * exp(distance[i * N + j]) * hyperparam_der); + tile.data()[i * N + j] = + factor * distance.data()[i * N + j] * exp(distance.data()[i * N + j]) * hyperparam_der; } } return tile; @@ -178,9 +178,9 @@ double adam_step( ///////////////////////////////////////////////////////////////////////// // Loss -double compute_loss(const std::vector &K_diag_tile, - const std::vector &alpha_tile, - const std::vector &y_tile, +double compute_loss(std::span K_diag_tile, + std::span alpha_tile, + std::span y_tile, std::size_t N) { // l = y^T * alpha + \sum_i^N log(L_ii^2) @@ -196,7 +196,7 @@ double compute_loss(const std::vector &K_diag_tile, return l; } -double add_losses(const std::vector &losses, std::size_t N, std::size_t n_tiles) +double add_losses(std::span losses, std::size_t N, std::size_t n_tiles) { // 0.5 * \sum losses + const double l = 0.0; @@ -218,17 +218,17 @@ double compute_gradient(double trace, double dot, std::size_t N, std::size_t n_t return 0.5 / static_cast(N * n_tiles) * (trace - dot); } -double compute_trace(const std::vector &diagonal, double trace) +double compute_trace(std::span diagonal, double trace) { return trace + std::reduce(diagonal.begin(), diagonal.end()); } -double compute_dot(const std::vector &vector_T, const std::vector &vector, double result) +double compute_dot(std::span vector_T, std::span vector, double result) { return result + dot(vector_T, vector, static_cast(vector.size())); } -double compute_trace_diag(const std::vector &tile, double trace, std::size_t N) +double compute_trace_diag(std::span tile, double trace, std::size_t N) { double local_trace = 0.0; for (std::size_t i = 0; i < N; ++i) diff --git a/core/src/cpu/gp_uncertainty.cpp b/core/src/cpu/gp_uncertainty.cpp index a0cf4511..5f03366f 100644 --- a/core/src/cpu/gp_uncertainty.cpp +++ b/core/src/cpu/gp_uncertainty.cpp @@ -1,23 +1,20 @@ #include "gprat/cpu/gp_uncertainty.hpp" +#include "gprat/tile_data.hpp" + GPRAT_NS_BEGIN namespace cpu { -hpx::shared_future> get_matrix_diagonal(hpx::shared_future> f_A, std::size_t M) +mutable_tile_data get_matrix_diagonal(const const_tile_data &A, std::size_t M) { - auto A = f_A.get(); - // Preallocate memory - std::vector tile; - tile.reserve(M); - // Add elements + mutable_tile_data tile(M); for (std::size_t i = 0; i < M; ++i) { - tile.push_back(A[i * M + i]); + tile.data()[i] = A.data()[i * M + i]; } - - return hpx::make_ready_future(std::move(tile)); + return tile; } } // end of namespace cpu diff --git a/core/src/cpu/tiled_algorithms.cpp b/core/src/cpu/tiled_algorithms.cpp index 88c8a468..18b416c5 100644 --- a/core/src/cpu/tiled_algorithms.cpp +++ b/core/src/cpu/tiled_algorithms.cpp @@ -4,6 +4,7 @@ #include "gprat/cpu/gp_algorithms.hpp" #include "gprat/cpu/gp_optimizer.hpp" #include "gprat/cpu/gp_uncertainty.hpp" +#include "gprat/detail/async_helpers.hpp" #include @@ -19,33 +20,23 @@ void right_looking_cholesky_tiled(Tiled_matrix &ft_tiles, int N, std::size_t n_t for (std::size_t k = 0; k < n_tiles; k++) { // POTRF: Compute Cholesky factor L - ft_tiles[k * n_tiles + k] = - hpx::dataflow(hpx::annotated_function(potrf, "cholesky_tiled"), ft_tiles[k * n_tiles + k], N); + ft_tiles[k * n_tiles + k] = detail::named_dataflow("cholesky_tiled", ft_tiles[k * n_tiles + k], N); for (std::size_t m = k + 1; m < n_tiles; m++) { // TRSM: Solve X * L^T = A - ft_tiles[m * n_tiles + k] = hpx::dataflow( - hpx::annotated_function(trsm, "cholesky_tiled"), - ft_tiles[k * n_tiles + k], - ft_tiles[m * n_tiles + k], - N, - N, - Blas_trans, - Blas_right); + ft_tiles[m * n_tiles + k] = detail::named_dataflow( + "cholesky_tiled", ft_tiles[k * n_tiles + k], ft_tiles[m * n_tiles + k], N, N, Blas_trans, Blas_right); } for (std::size_t m = k + 1; m < n_tiles; m++) { // SYRK: A = A - B * B^T - ft_tiles[m * n_tiles + m] = hpx::dataflow( - hpx::annotated_function(syrk, "cholesky_tiled"), - ft_tiles[m * n_tiles + m], - ft_tiles[m * n_tiles + k], - N); + ft_tiles[m * n_tiles + m] = + detail::named_dataflow("cholesky_tiled", ft_tiles[m * n_tiles + m], ft_tiles[m * n_tiles + k], N); for (std::size_t n = k + 1; n < m; n++) { // GEMM: C = C - A * B^T - ft_tiles[m * n_tiles + n] = hpx::dataflow( - hpx::annotated_function(gemm, "cholesky_tiled"), + ft_tiles[m * n_tiles + n] = detail::named_dataflow( + "cholesky_tiled", ft_tiles[m * n_tiles + k], ft_tiles[n * n_tiles + k], ft_tiles[m * n_tiles + n], @@ -66,17 +57,13 @@ void forward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, int N, st for (std::size_t k = 0; k < n_tiles; k++) { // TRSM: Solve L * x = a - ft_rhs[k] = hpx::dataflow( - hpx::annotated_function(trsv, "triangular_solve_tiled"), - ft_tiles[k * n_tiles + k], - ft_rhs[k], - N, - Blas_no_trans); + ft_rhs[k] = detail::named_dataflow( + "triangular_solve_tiled", ft_tiles[k * n_tiles + k], ft_rhs[k], N, Blas_no_trans); for (std::size_t m = k + 1; m < n_tiles; m++) { // GEMV: b = b - A * a - ft_rhs[m] = hpx::dataflow( - hpx::annotated_function(gemv, "triangular_solve_tiled"), + ft_rhs[m] = detail::named_dataflow( + "triangular_solve_tiled", ft_tiles[m * n_tiles + k], ft_rhs[k], ft_rhs[m], @@ -94,18 +81,14 @@ void backward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, int N, s { std::size_t k = static_cast(k_); // TRSM: Solve L^T * x = a - ft_rhs[k] = hpx::dataflow( - hpx::annotated_function(trsv, "triangular_solve_tiled"), - ft_tiles[k * n_tiles + k], - ft_rhs[k], - N, - Blas_trans); + ft_rhs[k] = + detail::named_dataflow("triangular_solve_tiled", ft_tiles[k * n_tiles + k], ft_rhs[k], N, Blas_trans); for (int m_ = k_ - 1; m_ >= 0; m_--) // int instead of std::size_t for last comparison { std::size_t m = static_cast(m_); // GEMV:b = b - A^T * a - ft_rhs[m] = hpx::dataflow( - hpx::annotated_function(gemv, "triangular_solve_tiled"), + ft_rhs[m] = detail::named_dataflow( + "triangular_solve_tiled", ft_tiles[k * n_tiles + m], ft_rhs[k], ft_rhs[m], @@ -125,8 +108,8 @@ void forward_solve_tiled_matrix( for (std::size_t k = 0; k < n_tiles; k++) { // TRSM: solve L * X = A - ft_rhs[k * m_tiles + c] = hpx::dataflow( - hpx::annotated_function(trsm, "triangular_solve_tiled_matrix"), + ft_rhs[k * m_tiles + c] = detail::named_dataflow( + "triangular_solve_tiled_matrix", ft_tiles[k * n_tiles + k], ft_rhs[k * m_tiles + c], N, @@ -136,8 +119,8 @@ void forward_solve_tiled_matrix( for (std::size_t m = k + 1; m < n_tiles; m++) { // GEMM: C = C - A * B - ft_rhs[m * m_tiles + c] = hpx::dataflow( - hpx::annotated_function(gemm, "triangular_solve_tiled_matrix"), + ft_rhs[m * m_tiles + c] = detail::named_dataflow( + "triangular_solve_tiled_matrix", ft_tiles[m * n_tiles + k], ft_rhs[k * m_tiles + c], ft_rhs[m * m_tiles + c], @@ -160,8 +143,8 @@ void backward_solve_tiled_matrix( { std::size_t k = static_cast(k_); // TRSM: solve L^T * X = A - ft_rhs[k * m_tiles + c] = hpx::dataflow( - hpx::annotated_function(trsm, "triangular_solve_tiled_matrix"), + ft_rhs[k * m_tiles + c] = detail::named_dataflow( + "triangular_solve_tiled_matrix", ft_tiles[k * n_tiles + k], ft_rhs[k * m_tiles + c], N, @@ -172,8 +155,8 @@ void backward_solve_tiled_matrix( { std::size_t m = static_cast(m_); // GEMM: C = C - A^T * B - ft_rhs[m * m_tiles + c] = hpx::dataflow( - hpx::annotated_function(gemm, "triangular_solve_tiled_matrix"), + ft_rhs[m * m_tiles + c] = detail::named_dataflow( + "triangular_solve_tiled_matrix", ft_tiles[k * n_tiles + m], ft_rhs[k * m_tiles + c], ft_rhs[m * m_tiles + c], @@ -199,8 +182,8 @@ void matrix_vector_tiled(Tiled_matrix &ft_tiles, { for (std::size_t m = 0; m < n_tiles; m++) { - ft_rhs[k] = hpx::dataflow( - hpx::annotated_function(gemv, "prediction_tiled"), + ft_rhs[k] = detail::named_dataflow( + "prediction_tiled", ft_tiles[k * n_tiles + m], ft_vector[m], ft_rhs[k], @@ -220,12 +203,8 @@ void symmetric_matrix_matrix_diagonal_tiled( for (std::size_t n = 0; n < n_tiles; ++n) { // Compute inner product to obtain diagonal elements of // V^T * V <=> cross(K) * K^-1 * cross(K)^T - ft_vector[i] = hpx::dataflow( - hpx::annotated_function(dot_diag_syrk, "posterior_tiled"), - ft_tiles[n * m_tiles + i], - ft_vector[i], - N, - M); + ft_vector[i] = + detail::named_dataflow("posterior_tiled", ft_tiles[n * m_tiles + i], ft_vector[i], N, M); } } } @@ -241,8 +220,8 @@ void symmetric_matrix_matrix_tiled( { // (SYRK for (c == k) possible) // GEMM: C = C - A^T * B - ft_result[c * m_tiles + k] = hpx::dataflow( - hpx::annotated_function(&gemm, "triangular_solve_tiled_matrix"), + ft_result[c * m_tiles + k] = detail::named_dataflow( + "triangular_solve_tiled_matrix", ft_tiles[m * m_tiles + c], ft_tiles[m * m_tiles + k], ft_result[c * m_tiles + k], @@ -260,8 +239,7 @@ void vector_difference_tiled(Tiled_vector &ft_minuend, Tiled_vector &ft_subtrahe { for (std::size_t i = 0; i < m_tiles; i++) { - ft_subtrahend[i] = - hpx::dataflow(hpx::annotated_function(&axpy, "uncertainty_tiled"), ft_minuend[i], ft_subtrahend[i], M); + ft_subtrahend[i] = detail::named_dataflow("uncertainty_tiled", ft_minuend[i], ft_subtrahend[i], M); } } @@ -269,8 +247,7 @@ void matrix_diagonal_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_vector, int { for (std::size_t i = 0; i < m_tiles; i++) { - ft_vector[i] = hpx::dataflow( - hpx::annotated_function(get_matrix_diagonal, "uncertainty_tiled"), ft_tiles[i * m_tiles + i], M); + ft_vector[i] = detail::named_dataflow("uncertainty_tiled", ft_tiles[i * m_tiles + i], M); } } @@ -285,15 +262,11 @@ void compute_loss_tiled(Tiled_matrix &ft_tiles, loss_tiled.reserve(n_tiles); for (std::size_t k = 0; k < n_tiles; k++) { - loss_tiled.push_back(hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&compute_loss), "loss_tiled"), - ft_tiles[k * n_tiles + k], - ft_alpha[k], - ft_y[k], - N)); + loss_tiled.push_back( + detail::named_dataflow("loss_tiled", ft_tiles[k * n_tiles + k], ft_alpha[k], ft_y[k], N)); } - loss = hpx::dataflow(hpx::annotated_function(hpx::unwrapping(&add_losses), "loss_tiled"), loss_tiled, N, n_tiles); + loss = detail::named_dataflow("loss_tiled", loss_tiled, N, n_tiles); } void update_hyperparameter_tiled( @@ -336,8 +309,8 @@ void update_hyperparameter_tiled( // Asynchrnonous initialization for (std::size_t d = 0; d < n_tiles; d++) { - diag_tiles.push_back(hpx::async(hpx::annotated_function(gen_tile_zeros, "assemble"), N)); - inter_alpha.push_back(hpx::async(hpx::annotated_function(gen_tile_zeros, "assemble"), N)); + diag_tiles.push_back(detail::named_async("assemble", N)); + inter_alpha.push_back(detail::named_async("assemble", N)); } //////////////////////////////////// @@ -348,20 +321,14 @@ void update_hyperparameter_tiled( { for (std::size_t j = 0; j < n_tiles; ++j) { - diag_tiles[i] = hpx::dataflow( - hpx::annotated_function(dot_diag_gemm, "trace"), - ft_invK[i * n_tiles + j], - ft_gradK_param[j * n_tiles + i], - diag_tiles[i], - N, - N); + diag_tiles[i] = detail::named_dataflow( + "trace", ft_invK[i * n_tiles + j], ft_gradK_param[j * n_tiles + i], diag_tiles[i], N, N); } } // Compute the trace of the diagonal tiles for (std::size_t j = 0; j < n_tiles; ++j) { - trace = - hpx::dataflow(hpx::annotated_function(hpx::unwrapping(&compute_trace), "trace"), diag_tiles[j], trace); + trace = detail::named_dataflow("trace", diag_tiles[j], trace); } // Not sure if can be done this way // Step 2: Compute alpha^T * grad(K)_param * alpha (with alpha = inv(K) * y) @@ -370,8 +337,8 @@ void update_hyperparameter_tiled( { for (std::size_t m = 0; m < n_tiles; m++) { - inter_alpha[k] = hpx::dataflow( - hpx::annotated_function(gemv, "gemv"), + inter_alpha[k] = detail::named_dataflow( + "gemv", ft_gradK_param[k * n_tiles + m], ft_alpha[m], inter_alpha[k], @@ -384,10 +351,7 @@ void update_hyperparameter_tiled( // Compute alpha^T * inter_alpha for (std::size_t j = 0; j < n_tiles; ++j) { - dot = hpx::dataflow(hpx::annotated_function(hpx::unwrapping(&compute_dot), "grad_right_tiled"), - inter_alpha[j], - ft_alpha[j], - dot); + dot = detail::named_dataflow("grad_right_tiled", inter_alpha[j], ft_alpha[j], dot); } } else if (param_idx == 2) // @2: noise_variance @@ -398,19 +362,13 @@ void update_hyperparameter_tiled( // Step 1: Compute the trace of inv(K) * noise_variance for (std::size_t j = 0; j < n_tiles; ++j) { - trace = hpx::dataflow(hpx::annotated_function(hpx::unwrapping(&compute_trace_diag), "grad_left_tiled"), - ft_invK[j * n_tiles + j], - trace, - N); + trace = detail::named_dataflow("grad_left_tiled", ft_invK[j * n_tiles + j], trace, N); } //////////////////////////////////// // Step 2: Compute the alpha^T * alpha * noise_variance for (std::size_t j = 0; j < n_tiles; ++j) { - dot = hpx::dataflow(hpx::annotated_function(hpx::unwrapping(&compute_dot), "grad_right_tiled"), - ft_alpha[j], - ft_alpha[j], - dot); + dot = detail::named_dataflow("grad_right_tiled", ft_alpha[j], ft_alpha[j], dot); } factor = compute_sigmoid(to_unconstrained(sek_params.noise_variance, true)); @@ -423,10 +381,7 @@ void update_hyperparameter_tiled( // Compute gradient = trace + dot double gradient = - factor - * hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&compute_gradient), "update_hyperparam"), trace, dot, N, n_tiles) - .get(); + factor * detail::named_dataflow("update_hyperparam", trace, dot, N, n_tiles).get(); //////////////////////////////////// // PART 2: Update parameter diff --git a/core/src/gprat.cpp b/core/src/gprat.cpp index 9eb199cf..2ce13252 100644 --- a/core/src/gprat.cpp +++ b/core/src/gprat.cpp @@ -332,7 +332,7 @@ double GP::calculate_loss() .get(); } -std::vector> GP::cholesky() +std::vector> GP::cholesky() { return hpx::async( [this]() diff --git a/core/src/performance_counters.cpp b/core/src/performance_counters.cpp new file mode 100644 index 00000000..b363efa1 --- /dev/null +++ b/core/src/performance_counters.cpp @@ -0,0 +1,37 @@ +#include "gprat/performance_counters.hpp" + +#include +#include + +GPRAT_NS_BEGIN + +#define GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(name) \ + static std::atomic name(0); \ + std::uint64_t get_##name(bool reset) { return hpx::util::get_and_reset_value(name, reset); } + +GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_data_allocations) +GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_data_deallocations) + +#undef GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR + +void track_tile_data_allocation(std::size_t /*size*/) { tile_data_allocations += 1; } + +void track_tile_data_deallocation(std::size_t /*size*/) { tile_data_deallocations += 1; } + +void register_performance_counters() +{ + hpx::performance_counters::install_counter_type( + "/gprat/tile_data/num_allocations", + &get_tile_data_allocations, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_data/num_deallocations", + &get_tile_data_deallocations, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); +} + +GPRAT_NS_END diff --git a/examples/gprat_cpp/src/execute.cpp b/examples/gprat_cpp/src/execute.cpp index d9ea2230..97ce3d8f 100644 --- a/examples/gprat_cpp/src/execute.cpp +++ b/examples/gprat_cpp/src/execute.cpp @@ -98,30 +98,28 @@ int main(int argc, char *argv[]) // Measure the time taken to execute gp.cholesky(); auto start_cholesky = std::chrono::high_resolution_clock::now(); - std::vector> choleksy_cpu = gp_cpu.cholesky(); + const auto choleksy_cpu = gp_cpu.cholesky(); auto end_cholesky = std::chrono::high_resolution_clock::now(); cholesky_time = end_cholesky - start_cholesky; // Measure the time taken to execute gp.optimize(hpar); auto start_opt = std::chrono::high_resolution_clock::now(); - std::vector losses = gp_cpu.optimize(hpar); + const auto losses = gp_cpu.optimize(hpar); auto end_opt = std::chrono::high_resolution_clock::now(); opt_time = end_opt - start_opt; auto start_pred_uncer = std::chrono::high_resolution_clock::now(); - std::vector> sum_cpu = - gp_cpu.predict_with_uncertainty(test_input.data, result.first, result.second); + const auto sum_cpu = gp_cpu.predict_with_uncertainty(test_input.data, result.first, result.second); auto end_pred_uncer = std::chrono::high_resolution_clock::now(); pred_uncer_time = end_pred_uncer - start_pred_uncer; auto start_pred_full_cov = std::chrono::high_resolution_clock::now(); - std::vector> full_cpu = - gp_cpu.predict_with_full_cov(test_input.data, result.first, result.second); + const auto full_cpu = gp_cpu.predict_with_full_cov(test_input.data, result.first, result.second); auto end_pred_full_cov = std::chrono::high_resolution_clock::now(); pred_full_cov_time = end_pred_full_cov - start_pred_full_cov; auto start_pred = std::chrono::high_resolution_clock::now(); - std::vector pred_cpu = gp_cpu.predict(test_input.data, result.first, result.second); + const auto pred_cpu = gp_cpu.predict(test_input.data, result.first, result.second); auto end_pred = std::chrono::high_resolution_clock::now(); pred_time = end_pred - start_pred; } @@ -147,7 +145,7 @@ int main(int argc, char *argv[]) gprat::start_hpx_runtime(new_argc, new_argv); auto start_cholesky = std::chrono::high_resolution_clock::now(); - std::vector> choleksy_gpu = gp_gpu.cholesky(); + const auto choleksy_gpu = gp_gpu.cholesky(); auto end_cholesky = std::chrono::high_resolution_clock::now(); cholesky_time = end_cholesky - start_cholesky; @@ -155,19 +153,17 @@ int main(int argc, char *argv[]) opt_time = std::chrono::seconds(-1); auto start_pred_uncer = std::chrono::high_resolution_clock::now(); - std::vector> sum_gpu = - gp_gpu.predict_with_uncertainty(test_input.data, result.first, result.second); + const auto sum_gpu = gp_gpu.predict_with_uncertainty(test_input.data, result.first, result.second); auto end_pred_uncer = std::chrono::high_resolution_clock::now(); pred_uncer_time = end_pred_uncer - start_pred_uncer; auto start_pred_full_cov = std::chrono::high_resolution_clock::now(); - std::vector> full_gpu = - gp_gpu.predict_with_full_cov(test_input.data, result.first, result.second); + const auto full_gpu = gp_gpu.predict_with_full_cov(test_input.data, result.first, result.second); auto end_pred_full_cov = std::chrono::high_resolution_clock::now(); pred_full_cov_time = end_pred_full_cov - start_pred_full_cov; auto start_pred = std::chrono::high_resolution_clock::now(); - std::vector pred_gpu = gp_gpu.predict(test_input.data, result.first, result.second); + const auto pred_gpu = gp_gpu.predict(test_input.data, result.first, result.second); auto end_pred = std::chrono::high_resolution_clock::now(); pred_time = end_pred - start_pred; } diff --git a/test/src/output_correctness.cpp b/test/src/output_correctness.cpp index 250a3341..9ac5aa58 100644 --- a/test/src/output_correctness.cpp +++ b/test/src/output_correctness.cpp @@ -41,6 +41,36 @@ void tag_invoke(boost::json::value_from_tag, boost::json::value &jv, const gprat }; } +template +std::vector to_vector(const gprat::const_tile_data &data) +{ + return { data.begin(), data.end() }; +} + +template +std::vector> to_vector(const std::vector> &data) +{ + std::vector> out; + out.reserve(data.size()); + for (const auto &row : data) + { + out.emplace_back(to_vector(row)); + } + return out; +} + +template +std::vector> to_vector(const std::vector> &data) +{ + std::vector> out; + out.reserve(data.size()); + for (const auto &row : data) + { + out.emplace_back(to_vector(row)); + } + return out; +} + // This helper function deduces the type and assigns the value with the matching key template inline void extract(const boost::json::object &obj, T &t, std::string_view key) @@ -94,15 +124,10 @@ gprat_results run_on_data_cpu(const std::string &train_path, const std::string & gprat::start_hpx_runtime(0, nullptr); gprat_results results_cpu; - - results_cpu.choleksy = gp_cpu.cholesky(); - + results_cpu.choleksy = to_vector(gp_cpu.cholesky()); results_cpu.losses = gp_cpu.optimize(hpar); - results_cpu.sum = gp_cpu.predict_with_uncertainty(test_input.data, test_tiles.first, test_tiles.second); - results_cpu.full = gp_cpu.predict_with_full_cov(test_input.data, test_tiles.first, test_tiles.second); - results_cpu.pred = gp_cpu.predict(test_input.data, test_tiles.first, test_tiles.second); // Stop the HPX runtime @@ -143,7 +168,7 @@ gprat_results run_on_data_gpu(const std::string &train_path, const std::string & gprat::start_hpx_runtime(0, nullptr); gprat_results results_gpu; - results_gpu.choleksy = gp_gpu.cholesky(); + results_gpu.choleksy = to_vector(gp_gpu.cholesky()); // NOTE: optimize and optimize_step are currently not implemented for GPU results_gpu.sum_no_optimize = gp_gpu.predict_with_uncertainty(test_input.data, test_tiles.first, test_tiles.second); results_gpu.full_no_optimize = gp_gpu.predict_with_full_cov(test_input.data, test_tiles.first, test_tiles.second); From 1977eea930d92febb1758d6dad1974a729ae8031 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sun, 10 Aug 2025 21:00:57 +0200 Subject: [PATCH 13/56] feat(core): Add simple-to-use per-function performance counters Powered by HPX's performance counter library. Since this library is only built if networking != none, guard against it being missing. --- core/include/gprat/cpu/adapter_cblas_fp32.hpp | 19 ++-- core/include/gprat/cpu/adapter_cblas_fp64.hpp | 3 +- core/include/gprat/performance_counters.hpp | 97 +++++++++++++++---- core/src/cpu/adapter_cblas_fp32.cpp | 47 ++++++++- core/src/cpu/adapter_cblas_fp64.cpp | 47 ++++++++- core/src/performance_counters.cpp | 45 ++++++--- 6 files changed, 212 insertions(+), 46 deletions(-) diff --git a/core/include/gprat/cpu/adapter_cblas_fp32.hpp b/core/include/gprat/cpu/adapter_cblas_fp32.hpp index 54852fee..015646b5 100644 --- a/core/include/gprat/cpu/adapter_cblas_fp32.hpp +++ b/core/include/gprat/cpu/adapter_cblas_fp32.hpp @@ -6,8 +6,7 @@ #include "gprat/detail/config.hpp" #include "gprat/tile_data.hpp" -#include -#include +#include GPRAT_NS_BEGIN @@ -96,11 +95,11 @@ trsv(const const_tile_data &L, const mutable_tile_data &a, int N, /** * @brief FP32 General matrix-vector multiplication: b = b - A(^T) * a - * @param f_A update matrix - * @param f_a update vector - * @param f_b base vector + * @param A update matrix + * @param a update vector + * @param b base vector * @param N matrix dimension - * @param alpha add or substract update to base vector + * @param alpha add or subtract update to base vector * @param transpose_A transpose update matrix * @return updated vector f_b */ @@ -140,8 +139,8 @@ mutable_tile_data dot_diag_gemm( /** * @brief FP32 AXPY: y - x - * @param f_y left vector - * @param f_x right vector + * @param y left vector + * @param x right vector * @param N vector length * @return y - x */ @@ -149,8 +148,8 @@ mutable_tile_data axpy(const mutable_tile_data &y, const const_til /** * @brief FP32 Dot product: a * b - * @param f_a left vector - * @param f_b right vector + * @param a left vector + * @param b right vector * @param N vector length * @return f_a * f_b */ diff --git a/core/include/gprat/cpu/adapter_cblas_fp64.hpp b/core/include/gprat/cpu/adapter_cblas_fp64.hpp index 4527c8bd..c2dab5d7 100644 --- a/core/include/gprat/cpu/adapter_cblas_fp64.hpp +++ b/core/include/gprat/cpu/adapter_cblas_fp64.hpp @@ -6,8 +6,7 @@ #include "gprat/detail/config.hpp" #include "gprat/tile_data.hpp" -#include -#include +#include GPRAT_NS_BEGIN diff --git a/core/include/gprat/performance_counters.hpp b/core/include/gprat/performance_counters.hpp index 86c35c82..402cb710 100644 --- a/core/include/gprat/performance_counters.hpp +++ b/core/include/gprat/performance_counters.hpp @@ -1,20 +1,77 @@ -#ifndef GPRAT_PERFORMANCE_COUNTERS_HPP -#define GPRAT_PERFORMANCE_COUNTERS_HPP - -#pragma once - -#include "gprat/detail/config.hpp" - -#include -#include - -GPRAT_NS_BEGIN - -void track_tile_data_allocation(std::size_t size); -void track_tile_data_deallocation(std::size_t size); - -void register_performance_counters(); - -GPRAT_NS_END - -#endif +#ifndef GPRAT_PERFORMANCE_COUNTERS_HPP +#define GPRAT_PERFORMANCE_COUNTERS_HPP + +#pragma once + +#include "gprat/detail/config.hpp" + +#include +#include +#include +#include +#include +#include + +GPRAT_NS_BEGIN + +/// The following is a very simple way of defining per-function metrics by using the function itself as a template +/// parameter ensuring that each function receives exactly one instantiation. +template +struct function_performance_metrics +{ + /// Number of times the function was called + static std::atomic num_calls; + + /// Total wall-clock time elapsed inside the function + static std::atomic elapsed_ns; +}; + +template +/*static*/ std::atomic function_performance_metrics::num_calls(0); +template +/*static*/ std::atomic function_performance_metrics::elapsed_ns(0); + +/// @brief This RAII helper allows us to time a function's total wall-clock execution time with minimal code. +struct scoped_function_timer +{ + explicit scoped_function_timer(std::atomic &num_calls, std::atomic &in_total) : + total(in_total) + { + ++num_calls; + } + + ~scoped_function_timer() + { + const auto elapsed = timer.elapsed_nanoseconds(); + HPX_ASSERT(elapsed >= 0); + if (elapsed > 0) + { + total += static_cast(elapsed); + } + } + + std::atomic &total; + hpx::chrono::high_resolution_timer timer; +}; + +/// @brief Time the execution of the enclosing function from the current point to its end. +/// @param local_function The function key that we're collecting performance information for. Usually the enclosing +/// function. +#define GPRAT_TIME_FUNCTION(local_function) \ + scoped_function_timer _gprat_fn_timer(function_performance_metrics::num_calls, \ + function_performance_metrics::elapsed_ns) + +template +std::uint64_t get_and_reset_function_elapsed(bool reset) +{ + return hpx::util::get_and_reset_value(function_performance_metrics::elapsed_ns, reset); +} + +void track_tile_data_allocation(std::size_t size); +void track_tile_data_deallocation(std::size_t size); + +void register_performance_counters(); + +GPRAT_NS_END + +#endif diff --git a/core/src/cpu/adapter_cblas_fp32.cpp b/core/src/cpu/adapter_cblas_fp32.cpp index acb3585c..29c06ec2 100644 --- a/core/src/cpu/adapter_cblas_fp32.cpp +++ b/core/src/cpu/adapter_cblas_fp32.cpp @@ -1,5 +1,11 @@ #include "gprat/cpu/adapter_cblas_fp32.hpp" +#include "gprat/performance_counters.hpp" + +#ifdef HPX_HAVE_MODULE_PERFORMANCE_COUNTERS +#include +#endif + #ifdef GPRAT_ENABLE_MKL // MKL CBLAS and LAPACKE #include "mkl_cblas.h" @@ -15,6 +21,7 @@ GPRAT_NS_BEGIN mutable_tile_data potrf(const mutable_tile_data &A, const int N) { + GPRAT_TIME_FUNCTION(&potrf); // POTRF: in-place Cholesky decomposition of A // use spotrf2 recursive version for better stability LAPACKE_spotrf2(LAPACK_ROW_MAJOR, 'L', N, A.data(), N); @@ -29,8 +36,8 @@ trsm(const const_tile_data &L, const int M, const BLAS_TRANSPOSE transpose_L, const BLAS_SIDE side_L) - { + GPRAT_TIME_FUNCTION(&trsm); // TRSM constants const float alpha = 1.0; // TRSM: in-place solve L(^T) * X = A or X * L(^T) = A where L lower triangular @@ -52,6 +59,7 @@ trsm(const const_tile_data &L, mutable_tile_data syrk(const mutable_tile_data &A, const const_tile_data &B, const int N) { + GPRAT_TIME_FUNCTION(&syrk); // SYRK constants const float alpha = -1.0; const float beta = 1.0; @@ -71,6 +79,7 @@ gemm(const const_tile_data &A, const BLAS_TRANSPOSE transpose_A, const BLAS_TRANSPOSE transpose_B) { + GPRAT_TIME_FUNCTION(&gemm); // GEMM constants const float alpha = -1.0; const float beta = 1.0; @@ -99,6 +108,7 @@ gemm(const const_tile_data &A, mutable_tile_data trsv(const const_tile_data &L, const mutable_tile_data &a, const int N, const BLAS_TRANSPOSE transpose_L) { + GPRAT_TIME_FUNCTION(&trsv); // TRSV: In-place solve L(^T) * x = a where L lower triangular cblas_strsv(CblasRowMajor, CblasLower, @@ -122,6 +132,7 @@ gemv(const const_tile_data &A, const BLAS_ALPHA alpha, const BLAS_TRANSPOSE transpose_A) { + GPRAT_TIME_FUNCTION(&gemv); // GEMV constants // const float alpha = -1.0; const float beta = 1.0; @@ -146,6 +157,7 @@ gemv(const const_tile_data &A, mutable_tile_data dot_diag_syrk(const const_tile_data &A, const mutable_tile_data &r, const int N, const int M) { + GPRAT_TIME_FUNCTION(&dot_diag_syrk); auto r_p = r.data(); auto A_p = A.data(); // r = r + diag(A^T * A) @@ -164,6 +176,7 @@ dot_diag_gemm(const const_tile_data &A, const int N, const int M) { + GPRAT_TIME_FUNCTION(&dot_diag_gemm); auto r_p = r.data(); auto A_p = A.data(); auto B_p = B.data(); @@ -179,14 +192,46 @@ dot_diag_gemm(const const_tile_data &A, mutable_tile_data axpy(const mutable_tile_data &y, const const_tile_data &x, const int N) { + GPRAT_TIME_FUNCTION(&axpy); cblas_saxpy(N, -1.0, x.data(), 1, y.data(), 1); return y; } float dot(std::span a, std::span b, const int N) { + GPRAT_TIME_FUNCTION(&dot); // DOT: a * b return cblas_sdot(N, a.data(), 1, b.data(), 1); } +#ifdef HPX_HAVE_MODULE_PERFORMANCE_COUNTERS +namespace detail +{ +void register_fp32_performance_counters() +{ + // XXX: you can do this with templates, but it's quite a bit more complicated +#define GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(name, fn_expr) \ + hpx::performance_counters::install_counter_type( \ + name, \ + get_and_reset_function_elapsed, \ + #fn_expr, \ + "", \ + hpx::performance_counters::counter_type::monotonically_increasing) + + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/potrf32/time", &potrf); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/trsm32/time", &trsm); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/syrk32/time", &syrk); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/gemm32/time", &gemm); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/trsv32/time", &trsv); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/gemv32/time", &gemv); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot_diag_syrk32/time", &dot_diag_syrk); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot_diag_gemm32/time", &dot_diag_gemm); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/axpy32/time", &axpy); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot32/time", &dot); + +#undef GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR +} +} // namespace detail +#endif + GPRAT_NS_END diff --git a/core/src/cpu/adapter_cblas_fp64.cpp b/core/src/cpu/adapter_cblas_fp64.cpp index aeedb1c3..46cb5c3f 100644 --- a/core/src/cpu/adapter_cblas_fp64.cpp +++ b/core/src/cpu/adapter_cblas_fp64.cpp @@ -1,5 +1,11 @@ #include "gprat/cpu/adapter_cblas_fp64.hpp" +#include "gprat/performance_counters.hpp" + +#ifdef HPX_HAVE_MODULE_PERFORMANCE_COUNTERS +#include +#endif + #ifdef GPRAT_ENABLE_MKL // MKL CBLAS and LAPACKE #include "mkl_cblas.h" @@ -15,6 +21,7 @@ GPRAT_NS_BEGIN mutable_tile_data potrf(const mutable_tile_data &A, const int N) { + GPRAT_TIME_FUNCTION(&potrf); // POTRF: in-place Cholesky decomposition of A // use dpotrf2 recursive version for better stability LAPACKE_dpotrf2(LAPACK_ROW_MAJOR, 'L', N, A.data(), N); @@ -29,8 +36,8 @@ trsm(const const_tile_data &L, const int M, const BLAS_TRANSPOSE transpose_L, const BLAS_SIDE side_L) - { + GPRAT_TIME_FUNCTION(&trsm); // TRSM constants const double alpha = 1.0; // TRSM: in-place solve L(^T) * X = A or X * L(^T) = A where L lower triangular @@ -53,6 +60,7 @@ trsm(const const_tile_data &L, mutable_tile_data syrk(const mutable_tile_data &A, const const_tile_data &B, const int N) { + GPRAT_TIME_FUNCTION(&syrk); // SYRK constants const double alpha = -1.0; const double beta = 1.0; @@ -72,6 +80,7 @@ gemm(const const_tile_data &A, const BLAS_TRANSPOSE transpose_A, const BLAS_TRANSPOSE transpose_B) { + GPRAT_TIME_FUNCTION(&gemm); // GEMM constants const double alpha = -1.0; const double beta = 1.0; @@ -100,6 +109,7 @@ gemm(const const_tile_data &A, mutable_tile_data trsv( const const_tile_data &L, const mutable_tile_data &a, const int N, const BLAS_TRANSPOSE transpose_L) { + GPRAT_TIME_FUNCTION(&trsv); // TRSV: In-place solve L(^T) * x = a where L lower triangular cblas_dtrsv(CblasRowMajor, CblasLower, @@ -123,6 +133,7 @@ gemv(const const_tile_data &A, const BLAS_ALPHA alpha, const BLAS_TRANSPOSE transpose_A) { + GPRAT_TIME_FUNCTION(&gemv); // GEMV constants // const double alpha = -1.0; const double beta = 1.0; @@ -147,6 +158,7 @@ gemv(const const_tile_data &A, mutable_tile_data dot_diag_syrk(const const_tile_data &A, const mutable_tile_data &r, const int N, const int M) { + GPRAT_TIME_FUNCTION(&dot_diag_syrk); auto r_p = r.data(); auto A_p = A.data(); // r = r + diag(A^T * A) @@ -165,6 +177,7 @@ dot_diag_gemm(const const_tile_data &A, const int N, const int M) { + GPRAT_TIME_FUNCTION(&dot_diag_gemm); auto r_p = r.data(); auto A_p = A.data(); auto B_p = B.data(); @@ -180,14 +193,46 @@ dot_diag_gemm(const const_tile_data &A, mutable_tile_data axpy(const mutable_tile_data &y, const const_tile_data &x, const int N) { + GPRAT_TIME_FUNCTION(&axpy); cblas_daxpy(N, -1.0, x.data(), 1, y.data(), 1); return y; } double dot(std::span a, std::span b, const int N) { + GPRAT_TIME_FUNCTION(&dot); // DOT: a * b return cblas_ddot(N, a.data(), 1, b.data(), 1); } +#ifdef HPX_HAVE_MODULE_PERFORMANCE_COUNTERS +namespace detail +{ +void register_fp64_performance_counters() +{ + // XXX: you can do this with templates, but it's quite a bit more complicated +#define GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(name, fn_expr) \ + hpx::performance_counters::install_counter_type( \ + name, \ + get_and_reset_function_elapsed, \ + #fn_expr, \ + "", \ + hpx::performance_counters::counter_type::monotonically_increasing) + + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/potrf64/time", &potrf); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/trsm64/time", &trsm); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/syrk64/time", &syrk); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/gemm64/time", &gemm); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/trsv64/time", &trsv); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/gemv64/time", &gemv); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot_diag_syrk64/time", &dot_diag_syrk); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot_diag_gemm64/time", &dot_diag_gemm); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/axpy64/time", &axpy); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot64/time", &dot); + +#undef GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR +} +} // namespace detail +#endif + GPRAT_NS_END diff --git a/core/src/performance_counters.cpp b/core/src/performance_counters.cpp index b363efa1..c405cff4 100644 --- a/core/src/performance_counters.cpp +++ b/core/src/performance_counters.cpp @@ -1,7 +1,10 @@ #include "gprat/performance_counters.hpp" #include +#include +#ifdef HPX_HAVE_MODULE_PERFORMANCE_COUNTERS #include +#endif GPRAT_NS_BEGIN @@ -18,20 +21,38 @@ void track_tile_data_allocation(std::size_t /*size*/) { tile_data_allocations += void track_tile_data_deallocation(std::size_t /*size*/) { tile_data_deallocations += 1; } +#ifdef HPX_HAVE_MODULE_PERFORMANCE_COUNTERS +// These are non-public functions of their respective CUs. +namespace detail +{ +void register_fp32_performance_counters(); +void register_fp64_performance_counters(); +} // namespace detail + +void register_performance_counters() +{ + // XXX: you can do this with templates, but it's quite a bit more complicated +#define GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(name, stats_expr) \ + hpx::performance_counters::install_counter_type( \ + name, \ + [](bool reset) { return hpx::util::get_and_reset_value(stats_expr, reset); }, \ + #stats_expr, \ + "", \ + hpx::performance_counters::counter_type::monotonically_increasing) + + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/tile_data/num_allocations", tile_data_allocations); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/tile_data/num_deallocations", tile_data_deallocations); + +#undef GPRAT_MAKE_STATISTICS_ACCESSOR + + detail::register_fp32_performance_counters(); + detail::register_fp64_performance_counters(); +} +#else void register_performance_counters() { - hpx::performance_counters::install_counter_type( - "/gprat/tile_data/num_allocations", - &get_tile_data_allocations, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_data/num_deallocations", - &get_tile_data_deallocations, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); + // no-op for binary compatibility } +#endif GPRAT_NS_END From 928b2692f2295e30225ccbc218f8a81de79b5956 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Fri, 25 Jul 2025 23:25:44 +0200 Subject: [PATCH 14/56] feat(core): Use NUMA-aware allocator for tile data --- core/CMakeLists.txt | 1 + core/include/gprat/tile_data.hpp | 73 +++++++++++++++++++++++--------- core/src/tile_data.cpp | 34 +++++++++++++++ 3 files changed, 89 insertions(+), 19 deletions(-) create mode 100644 core/src/tile_data.cpp diff --git a/core/CMakeLists.txt b/core/CMakeLists.txt index 8309eba2..38f472ea 100644 --- a/core/CMakeLists.txt +++ b/core/CMakeLists.txt @@ -13,6 +13,7 @@ set(SOURCE_FILES src/utils.cpp src/performance_counters.cpp src/target.cpp + src/tile_data.cpp src/kernels.cpp src/hyperparameters.cpp src/cpu/gp_functions.cpp diff --git a/core/include/gprat/tile_data.hpp b/core/include/gprat/tile_data.hpp index 39d48dd9..006ac62b 100644 --- a/core/include/gprat/tile_data.hpp +++ b/core/include/gprat/tile_data.hpp @@ -4,13 +4,59 @@ #pragma once #include "gprat/detail/config.hpp" -#include "gprat/performance_counters.hpp" #include #include GPRAT_NS_BEGIN +namespace detail +{ +void *allocate_tile_data(std::size_t num_bytes); +void deallocate_tile_data(void *p, std::size_t num_bytes); + +template +struct tile_data_allocator +{ + typedef T value_type; + + tile_data_allocator() = default; + + template + constexpr tile_data_allocator(const tile_data_allocator &) noexcept + { } + + [[nodiscard]] T *allocate(std::size_t n) + { + if (n > (std::numeric_limits::max)() / sizeof(T)) + { + throw std::bad_array_new_length(); + } + + if (auto p = static_cast(allocate_tile_data(n * sizeof(T)))) + { + return p; + } + + throw std::bad_alloc(); + } + + void deallocate(T *p, std::size_t n) noexcept { deallocate_tile_data(p, n * sizeof(T)); } +}; + +template +bool operator==(const tile_data_allocator &, const tile_data_allocator &) +{ + return true; +} + +template +bool operator!=(const tile_data_allocator &, const tile_data_allocator &) +{ + return false; +} +} // namespace detail + /** * @brief Non-mutable reference-counted dynamic array of a given type T. * This class represents a simple reference-counted non-resizeable buffer with elements of type T. @@ -25,7 +71,7 @@ template class const_tile_data { protected: - typedef hpx::serialization::serialize_buffer cpu_buffer_type; + typedef hpx::serialization::serialize_buffer> cpu_buffer_type; struct hold_reference { @@ -38,25 +84,12 @@ class const_tile_data cpu_buffer_type data_; }; - // In case we want pooling down the road... - static T *allocate(std::size_t n) - { - track_tile_data_allocation(n); - return new T[n]; - } - - static void deallocate(T *p) noexcept - { - track_tile_data_deallocation(0); // we don't know here - delete[] p; - } - public: const_tile_data() = default; // Create a new (uninitialized) tile_data of the given size. explicit const_tile_data(std::size_t size) : - cpu_data_(allocate(size), size, cpu_buffer_type::take, &const_tile_data::deallocate) + cpu_data_(size) { } // Create a tile_data which acts as a proxy to a part of the embedded array. @@ -85,10 +118,12 @@ class const_tile_data return { cpu_data_.data(), cpu_data_.size() }; } + friend bool operator==(const const_tile_data &a, const const_tile_data &b) noexcept + { + return a.cpu_data_ == b.cpu_data_; + } + protected: - // Serialization support: even if all of the code below runs on one - // locality only, we need to provide an (empty) implementation for the - // serialization as all arguments passed to actions have to support this. friend class hpx::serialization::access; template diff --git a/core/src/tile_data.cpp b/core/src/tile_data.cpp new file mode 100644 index 00000000..24ef9eb3 --- /dev/null +++ b/core/src/tile_data.cpp @@ -0,0 +1,34 @@ +#include "gprat/tile_data.hpp" + +#include "gprat/performance_counters.hpp" + +#include + +GPRAT_NS_BEGIN + +namespace detail +{ + +void *allocate_tile_data(std::size_t num_bytes) +{ + auto &topology = hpx::get_runtime().get_topology(); + const auto bitmap = topology.cpuset_to_nodeset(topology.get_machine_affinity_mask()); + + track_tile_data_allocation(num_bytes); + return topology.allocate_membind(num_bytes, bitmap, hpx::threads::hpx_hwloc_membind_policy::membind_firsttouch, 0); +} + +void deallocate_tile_data(void *p, std::size_t num_bytes) +{ + track_tile_data_deallocation(num_bytes); + + if (hpx::is_running()) + { + auto &topology = hpx::get_runtime().get_topology(); + topology.deallocate(p, num_bytes); + } +} + +} // namespace detail + +GPRAT_NS_END From 3df5ebdc2c9f7ff2060aa1775197edff4c3a79b7 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Mon, 11 Aug 2025 22:45:00 +0200 Subject: [PATCH 15/56] chore(core): Consistently use std::size_t Quite a few functions took `int` parameters just to cast them to `std::size_t` everywhere. --- core/include/gprat/cpu/gp_functions.hpp | 42 +- core/include/gprat/cpu/tiled_algorithms.hpp | 54 ++- core/include/gprat/gprat.hpp | 46 +-- core/include/gprat/hyperparameters.hpp | 6 +- core/include/gprat/utils.hpp | 11 +- core/src/cpu/gp_algorithms.cpp | 2 +- core/src/cpu/gp_functions.cpp | 420 +++++++------------- core/src/cpu/tiled_algorithms.cpp | 49 ++- core/src/gprat.cpp | 56 ++- core/src/hyperparameters.cpp | 2 +- core/src/utils.cpp | 25 +- examples/gprat_cpp/src/execute.cpp | 8 +- test/src/output_correctness.cpp | 4 +- 13 files changed, 320 insertions(+), 405 deletions(-) diff --git a/core/include/gprat/cpu/gp_functions.hpp b/core/include/gprat/cpu/gp_functions.hpp index 11a61617..1df7607b 100644 --- a/core/include/gprat/cpu/gp_functions.hpp +++ b/core/include/gprat/cpu/gp_functions.hpp @@ -30,9 +30,9 @@ namespace cpu std::vector> cholesky(const std::vector &training_input, const SEKParams &sek_params, - int n_tiles, - int n_tile_size, - int n_regressors); + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors); /** * @brief Compute the predictions without uncertainties. @@ -54,11 +54,11 @@ predict(const std::vector &training_input, const std::vector &training_output, const std::vector &test_input, const SEKParams &sek_params, - int n_tiles, - int n_tile_size, - int m_tiles, - int m_tile_size, - int n_regressors); + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t m_tiles, + std::size_t m_tile_size, + std::size_t n_regressors); /** * @brief Compute the predictions with uncertainties. @@ -80,11 +80,11 @@ std::vector> predict_with_uncertainty( const std::vector &training_output, const std::vector &test_input, const SEKParams &sek_params, - int n_tiles, - int n_tile_size, - int m_tiles, - int m_tile_size, - int n_regressors); + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t m_tiles, + std::size_t m_tile_size, + std::size_t n_regressors); /** * @brief Compute the predictions with full covariance matrix. @@ -106,11 +106,11 @@ std::vector> predict_with_full_cov( const std::vector &training_output, const std::vector &test_data, const SEKParams &sek_params, - int n_tiles, - int n_tile_size, - int m_tiles, - int m_tile_size, - int n_regressors); + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t m_tiles, + std::size_t m_tile_size, + std::size_t n_regressors); /** * @brief Compute loss for given data and Gaussian process model @@ -127,9 +127,9 @@ std::vector> predict_with_full_cov( double compute_loss(const std::vector &training_input, const std::vector &training_output, const SEKParams &sek_params, - int n_tiles, - int n_tile_size, - int n_regressors); + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors); /** * @brief Perform optimization for a given number of iterations diff --git a/core/include/gprat/cpu/tiled_algorithms.hpp b/core/include/gprat/cpu/tiled_algorithms.hpp index 0f297b1b..a8706fe9 100644 --- a/core/include/gprat/cpu/tiled_algorithms.hpp +++ b/core/include/gprat/cpu/tiled_algorithms.hpp @@ -28,7 +28,7 @@ namespace cpu * @param N Tile size per dimension. * @param n_tiles Number of tiles per dimension. */ -void right_looking_cholesky_tiled(Tiled_matrix &ft_tiles, int N, std::size_t n_tiles); +void right_looking_cholesky_tiled(Tiled_matrix &ft_tiles, std::size_t N, std::size_t n_tiles); // Tiled Triangular Solve Algorithms @@ -40,7 +40,7 @@ void right_looking_cholesky_tiled(Tiled_matrix &ft_tiles, int N, std::size_t n_t * @param N Tile size per dimension. * @param n_tiles Number of tiles per dimension. */ -void forward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, int N, std::size_t n_tiles); +void forward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, std::size_t N, std::size_t n_tiles); /** * @brief Perform tiled backward triangular matrix-vector solve. @@ -50,7 +50,7 @@ void forward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, int N, st * @param N Tile size per dimension. * @param n_tiles Number of tiles per dimension. */ -void backward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, int N, std::size_t n_tiles); +void backward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, std::size_t N, std::size_t n_tiles); /** * @brief Perform tiled forward triangular matrix-matrix solve. @@ -62,8 +62,12 @@ void backward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, int N, s * @param n_tiles Number of tiles in first dimension. * @param m_tiles Number of tiles in second dimension. */ -void forward_solve_tiled_matrix( - Tiled_matrix &ft_tiles, Tiled_matrix &ft_rhs, int N, int M, std::size_t n_tiles, std::size_t m_tiles); +void forward_solve_tiled_matrix(Tiled_matrix &ft_tiles, + Tiled_matrix &ft_rhs, + std::size_t N, + std::size_t M, + std::size_t n_tiles, + std::size_t m_tiles); /** * @brief Perform tiled backward triangular matrix-matrix solve. @@ -75,15 +79,19 @@ void forward_solve_tiled_matrix( * @param n_tiles Number of tiles in first dimension. * @param m_tiles Number of tiles in second dimension. */ -void backward_solve_tiled_matrix( - Tiled_matrix &ft_tiles, Tiled_matrix &ft_rhs, int N, int M, std::size_t n_tiles, std::size_t m_tiles); +void backward_solve_tiled_matrix(Tiled_matrix &ft_tiles, + Tiled_matrix &ft_rhs, + std::size_t N, + std::size_t M, + std::size_t n_tiles, + std::size_t m_tiles); /** * @brief Perform tiled matrix-vector multiplication * * @param ft_tiles Tiled matrix represented as a vector of futurized tiles. * @param ft_vector Tiled vector represented as a vector of futurized tiles. - * @param ft_rhsTiled solution represented as a vector of futurized tiles. + * @param ft_rhs Tiled solution represented as a vector of futurized tiles. * @param N_row Tile size of first dimension. * @param N_col Tile size of second dimension. * @param n_tiles Number of tiles in first dimension. @@ -92,8 +100,8 @@ void backward_solve_tiled_matrix( void matrix_vector_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_vector, Tiled_vector &ft_rhs, - int N_row, - int N_col, + std::size_t N_row, + std::size_t N_col, std::size_t n_tiles, std::size_t m_tiles); @@ -108,7 +116,12 @@ void matrix_vector_tiled(Tiled_matrix &ft_tiles, * @param m_tiles Number of tiles in second dimension. */ void symmetric_matrix_matrix_diagonal_tiled( - Tiled_matrix &ft_tiles, Tiled_vector &ft_vector, int N, int M, std::size_t n_tiles, std::size_t m_tiles); + Tiled_matrix &ft_tiles, + Tiled_vector &ft_vector, + std::size_t N, + std::size_t M, + std::size_t n_tiles, + std::size_t m_tiles); /** * @brief Perform tiled symmetric k-rank update (ft_tiles^T * ft_tiles) @@ -120,18 +133,21 @@ void symmetric_matrix_matrix_diagonal_tiled( * @param n_tiles Number of tiles in first dimension. * @param m_tiles Number of tiles in second dimension. */ -void symmetric_matrix_matrix_tiled( - Tiled_matrix &ft_tiles, Tiled_matrix &ft_result, int N, int M, std::size_t n_tiles, std::size_t m_tiles); +void symmetric_matrix_matrix_tiled(Tiled_matrix &ft_tiles, + Tiled_matrix &ft_result, + std::size_t N, + std::size_t M, + std::size_t n_tiles, + std::size_t m_tiles); /** * @brief Compute the difference between two tiled vectors * @param ft_minuend Tiled vector that is being subtracted from. * @param ft_subtrahend Tiled vector that is being subtracted. - * @param ft_difference Tiled vector that contains the result of the substraction. * @param M Tile size dimension. * @param m_tiles Number of tiles. */ -void vector_difference_tiled(Tiled_vector &ft_minuend, Tiled_vector &ft_substrahend, int M, std::size_t m_tiles); +void vector_difference_tiled(Tiled_vector &ft_minuend, Tiled_vector &ft_subtrahend, std::size_t M, std::size_t m_tiles); /** * @brief Extract the tiled diagonals of a tiled matrix @@ -140,7 +156,7 @@ void vector_difference_tiled(Tiled_vector &ft_minuend, Tiled_vector &ft_substrah * @param M Tile size per dimension. * @param m_tiles Number of tiles per dimension. */ -void matrix_diagonal_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_vector, int M, std::size_t m_tiles); +void matrix_diagonal_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_vector, std::size_t M, std::size_t m_tiles); /** * @brief Compute the negative log likelihood loss with a tiled covariance matrix K. @@ -158,14 +174,14 @@ void compute_loss_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_alpha, Tiled_vector &ft_y, hpx::shared_future &loss, - int N, + std::size_t N, std::size_t n_tiles); /** * @brief Updates a hyperparameter of the SEK kernel using Adam * * @param ft_invK Tiled inverse of the covariance matrix K represented as a vector of futurized tiles. - * @param ft_grad_param Tiled covariance matrix gradient w.r.t. a hyperparameter. + * @param ft_gradK_param Tiled covariance matrix gradient w.r.t. a hyperparameter. * @param ft_alpha Tiled vector containing the precomputed inv(K) * y where y is the training output. * @param adam_params Hyperparameter of the Adam optimizer * @param sek_params Hyperparameters of the SEK kernel @@ -180,7 +196,7 @@ void update_hyperparameter_tiled( const Tiled_vector &ft_alpha, const AdamParams &adam_params, SEKParams &sek_params, - int N, + std::size_t N, std::size_t n_tiles, std::size_t iter, std::size_t param_idx); diff --git a/core/include/gprat/gprat.hpp b/core/include/gprat/gprat.hpp index e41850b2..88c6972f 100644 --- a/core/include/gprat/gprat.hpp +++ b/core/include/gprat/gprat.hpp @@ -27,10 +27,10 @@ struct GP_data std::string file_path; /** @brief Number of samples in the data */ - int n_samples; + std::size_t n_samples; /** @brief Number of GP regressors */ - int n_regressors; + std::size_t n_regressors; /** @brief Vector containing the data */ std::vector data; @@ -41,10 +41,10 @@ struct GP_data * * The file specified by `f_path` must contain `n` samples. * - * @param f_path Path to the file + * @param file_path Path to the file * @param n Number of samples */ - GP_data(const std::string &file_path, int n, int n_reg); + GP_data(const std::string &file_path, std::size_t n, std::size_t n_reg); }; /** @@ -64,10 +64,10 @@ class GP std::vector training_output_; /** @brief Number of tiles */ - int n_tiles_; + std::size_t n_tiles_; /** @brief Size of each tile in each dimension */ - int n_tile_size_; + std::size_t n_tile_size_; /** * @brief List of bools indicating trainable parameters: lengthscale, @@ -82,7 +82,7 @@ class GP public: /** @brief Number of regressors */ - int n_reg; + std::size_t n_reg; /** * @brief Hyperarameters of the squared exponential kernel @@ -105,10 +105,10 @@ class GP */ GP(std::vector input, std::vector output, - int n_tiles, - int n_tile_size, - int n_regressors, - std::vector kernel_hyperparams, + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors, + const std::vector &kernel_hyperparams, std::vector trainable_bool, std::shared_ptr target); @@ -127,10 +127,10 @@ class GP */ GP(std::vector input, std::vector output, - int n_tiles, - int n_tile_size, - int n_regressors, - std::vector kernel_hyperparams, + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors, + const std::vector &kernel_hyperparams, std::vector trainable_bool); /** @@ -150,10 +150,10 @@ class GP */ GP(std::vector input, std::vector output, - int n_tiles, - int n_tile_size, - int n_regressors, - std::vector kernel_hyperparams, + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors, + const std::vector &kernel_hyperparams, std::vector trainable_bool, int gpu_id, int n_streams); @@ -176,14 +176,14 @@ class GP /** * @brief Predict output for test input */ - std::vector predict(const std::vector &test_data, int m_tiles, int m_tile_size); + std::vector predict(const std::vector &test_data, std::size_t m_tiles, std::size_t m_tile_size); /** * @brief Predict output for test input and additionally provide * uncertainty for the predictions. */ std::vector> - predict_with_uncertainty(const std::vector &test_data, int m_tiles, int m_tile_size); + predict_with_uncertainty(const std::vector &test_data, std::size_t m_tiles, std::size_t m_tile_size); /** * @brief Predict output for test input and additionally compute full @@ -196,7 +196,7 @@ class GP * @return Full covariance matrix */ std::vector> - predict_with_full_cov(const std::vector &test_data, int m_tiles, int m_tile_size); + predict_with_full_cov(const std::vector &test_data, std::size_t m_tiles, std::size_t m_tile_size); /** * @brief Optimize hyperparameters @@ -217,7 +217,7 @@ class GP * * @return loss */ - double optimize_step(AdamParams &adam_params, int iter); + double optimize_step(AdamParams &adam_params, std::size_t iter); /** * @brief Calculate loss for given data and Gaussian process model diff --git a/core/include/gprat/hyperparameters.hpp b/core/include/gprat/hyperparameters.hpp index e81bdf03..c980bd74 100644 --- a/core/include/gprat/hyperparameters.hpp +++ b/core/include/gprat/hyperparameters.hpp @@ -38,7 +38,7 @@ struct AdamParams /** * @brief Number of optimization iterations */ - int opt_iter; + std::size_t opt_iter; /** * @brief Initialize hyperparameters @@ -48,10 +48,8 @@ struct AdamParams * @param b2 beta2 * @param eps epsilon * @param opt_i number of optimization iterationsgp op - * @param M_T_init initial values for first moment vector - * @param V_T_init initial values for second moment vector */ - AdamParams(double lr = 0.001, double b1 = 0.9, double b2 = 0.999, double eps = 1e-8, int opt_i = 0); + AdamParams(double lr = 0.001, double b1 = 0.9, double b2 = 0.999, double eps = 1e-8, std::size_t opt_i = 0); /** * @brief Returns a string representation of the hyperparameters diff --git a/core/include/gprat/utils.hpp b/core/include/gprat/utils.hpp index d269c91c..86a4ddd2 100644 --- a/core/include/gprat/utils.hpp +++ b/core/include/gprat/utils.hpp @@ -20,16 +20,16 @@ GPRAT_NS_BEGIN * @param n_samples Number of samples * @param n_tile_size Size of each tile */ -int compute_train_tiles(int n_samples, int n_tile_size); +std::size_t compute_train_tiles(std::size_t n_samples, std::size_t n_tile_size); /** * @brief Compute the number of tiles for training data, given the number of * samples and the size of each tile. * * @param n_samples Number of samples - * @param n_tile_size Size of each tile + * @param n_tiles Size of each tile */ -int compute_train_tile_size(int n_samples, int n_tiles); +std::size_t compute_train_tile_size(std::size_t n_samples, std::size_t n_tiles); /** * @brief Compute the number of test tiles and the size of a test tile. @@ -41,7 +41,8 @@ int compute_train_tile_size(int n_samples, int n_tiles); * @param n_tiles Number of tiles * @param n_tile_size Size of each tile */ -std::pair compute_test_tiles(int n_test, int n_tiles, int n_tile_size); +std::pair +compute_test_tiles(std::size_t n_test, std::size_t n_tiles, std::size_t n_tile_size); /** * @brief Load data from file @@ -49,7 +50,7 @@ std::pair compute_test_tiles(int n_test, int n_tiles, int n_tile_size) * @param file_path Path to the file * @param n_samples Number of samples to load */ -std::vector load_data(const std::string &file_path, int n_samples, int offset); +std::vector load_data(const std::string &file_path, std::size_t n_samples, std::size_t offset); /** * @brief Print a vector diff --git a/core/src/cpu/gp_algorithms.cpp b/core/src/cpu/gp_algorithms.cpp index b02dfe4e..c99b570a 100644 --- a/core/src/cpu/gp_algorithms.cpp +++ b/core/src/cpu/gp_algorithms.cpp @@ -147,7 +147,7 @@ mutable_tile_data gen_tile_transpose(std::size_t N_row, std::size_t N_co mutable_tile_data gen_tile_output(std::size_t row, std::size_t N, std::span output) { mutable_tile_data tile(N); - std::copy(output.begin() + (N * row), output.begin() + (N * (row + 1)), tile.data()); + std::copy(output.data() + (N * row), output.data() + (N * (row + 1)), tile.data()); return tile; } diff --git a/core/src/cpu/gp_functions.cpp b/core/src/cpu/gp_functions.cpp index 4b70c691..0e32eac7 100644 --- a/core/src/cpu/gp_functions.cpp +++ b/core/src/cpu/gp_functions.cpp @@ -17,41 +17,40 @@ namespace cpu std::vector> cholesky(const std::vector &training_input, const SEKParams &sek_params, - int n_tiles, - int n_tile_size, - int n_regressors) + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors) { std::vector> result; // Tiled future data structures Tiled_matrix K_tiles; // Tiled covariance matrix // Preallocate memory - result.resize(static_cast(n_tiles * n_tiles)); - K_tiles.resize(static_cast(n_tiles * n_tiles)); // No reserve because of triangular structure + result.resize(n_tiles * n_tiles); + K_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure /////////////////////////////////////////////////////////////////////////// // Launch asynchronous assembly - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { for (std::size_t j = 0; j <= i; j++) { - K_tiles[i * static_cast(n_tiles) + j] = detail::named_async( + K_tiles[i * n_tiles + j] = detail::named_async( "assemble_tiled_K", i, j, n_tile_size, n_regressors, sek_params, training_input); } } /////////////////////////////////////////////////////////////////////////// // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(K_tiles, n_tile_size, static_cast(n_tiles)); + right_looking_cholesky_tiled(K_tiles, n_tile_size, n_tiles); /////////////////////////////////////////////////////////////////////////// // Synchronize - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { for (std::size_t j = 0; j <= i; j++) { - result[i * static_cast(n_tiles) + j] = - K_tiles[i * static_cast(n_tiles) + j].get(); + result[i * n_tiles + j] = K_tiles[i * n_tiles + j].get(); } } return result; @@ -62,11 +61,11 @@ predict(const std::vector &training_input, const std::vector &training_output, const std::vector &test_input, const SEKParams &sek_params, - int n_tiles, - int n_tile_size, - int m_tiles, - int m_tile_size, - int n_regressors) + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t m_tiles, + std::size_t m_tile_size, + std::size_t n_regressors) { /* * Prediction: hat(y)_M = cross(K)_MxN * K^-1_NxN * y_N @@ -94,65 +93,59 @@ predict(const std::vector &training_input, // Preallocate memory prediction_result.reserve(test_input.size()); - K_tiles.resize(static_cast(n_tiles * n_tiles)); // No reserve because of triangular structure - alpha_tiles.reserve(static_cast(n_tiles)); - cross_covariance_tiles.reserve(static_cast(m_tiles) * static_cast(n_tiles)); - prediction_tiles.reserve(static_cast(m_tiles)); + K_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure + alpha_tiles.reserve(n_tiles); + cross_covariance_tiles.reserve(m_tiles * n_tiles); + prediction_tiles.reserve(m_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous assembly - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { for (std::size_t j = 0; j <= i; j++) { - K_tiles[i * static_cast(n_tiles) + j] = detail::named_async( + K_tiles[i * n_tiles + j] = detail::named_async( "assemble_tiled_K", i, j, n_tile_size, n_regressors, sek_params, training_input); } } - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { alpha_tiles.push_back( detail::named_async("assemble_tiled_alpha", i, n_tile_size, training_output)); } - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { - for (std::size_t j = 0; j < static_cast(n_tiles); j++) + for (std::size_t j = 0; j < n_tiles; j++) { cross_covariance_tiles.push_back(detail::named_async( "assemble_pred", i, j, m_tile_size, n_tile_size, n_regressors, sek_params, test_input, training_input)); } } - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { prediction_tiles.push_back(detail::named_async("assemble_tiled", m_tile_size)); } /////////////////////////////////////////////////////////////////////////// // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(K_tiles, n_tile_size, static_cast(n_tiles)); + right_looking_cholesky_tiled(K_tiles, n_tile_size, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous triangular solve L * (L^T * alpha) = y - forward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, static_cast(n_tiles)); - backward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, static_cast(n_tiles)); + forward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); + backward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous prediction computation solve: \hat{y} = K_cross_cov * alpha matrix_vector_tiled( - cross_covariance_tiles, - alpha_tiles, - prediction_tiles, - m_tile_size, - n_tile_size, - static_cast(n_tiles), - static_cast(m_tiles)); + cross_covariance_tiles, alpha_tiles, prediction_tiles, m_tile_size, n_tile_size, n_tiles, m_tiles); /////////////////////////////////////////////////////////////////////////// // Synchronize prediction - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { auto tile = prediction_tiles[i].get(); std::copy_n(tile.data(), tile.size(), std::back_inserter(prediction_result)); @@ -165,11 +158,11 @@ std::vector> predict_with_uncertainty( const std::vector &training_output, const std::vector &test_input, const SEKParams &sek_params, - int n_tiles, - int n_tile_size, - int m_tiles, - int m_tile_size, - int n_regressors) + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t m_tiles, + std::size_t m_tile_size, + std::size_t n_regressors) { /* * Prediction: hat(y) = cross(K) * K^-1 * y @@ -210,139 +203,104 @@ std::vector> predict_with_uncertainty( prediction_result.reserve(test_input.size()); uncertainty_result.reserve(test_input.size()); - K_tiles.resize(static_cast(n_tiles * n_tiles)); // No reserve because of triangular structure - cross_covariance_tiles.reserve(static_cast(m_tiles) * static_cast(n_tiles)); - prediction_tiles.reserve(static_cast(m_tiles)); - alpha_tiles.reserve(static_cast(n_tiles)); + K_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure + cross_covariance_tiles.reserve(m_tiles * n_tiles); + prediction_tiles.reserve(m_tiles); + alpha_tiles.reserve(n_tiles); - t_cross_covariance_tiles.reserve(static_cast(n_tiles) * static_cast(m_tiles)); - prior_K_tiles.reserve(static_cast(m_tiles)); - uncertainty_tiles.reserve(static_cast(m_tiles)); + t_cross_covariance_tiles.reserve(n_tiles * m_tiles); + prior_K_tiles.reserve(m_tiles); + uncertainty_tiles.reserve(m_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous assembly - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { for (std::size_t j = 0; j <= i; j++) { - K_tiles[i * static_cast(n_tiles) + j] = detail::named_async( + K_tiles[i * n_tiles + j] = detail::named_async( "assemble_tiled_K", i, j, n_tile_size, n_regressors, sek_params, training_input); } } - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { - alpha_tiles.push_back(hpx::async( - hpx::annotated_function(gen_tile_output, "assemble_tiled_alpha"), i, n_tile_size, training_output)); + alpha_tiles.push_back( + detail::named_async("assemble_tiled_alpha", i, n_tile_size, training_output)); } - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { - for (std::size_t j = 0; j < static_cast(n_tiles); j++) + for (std::size_t j = 0; j < n_tiles; j++) { - cross_covariance_tiles.push_back(hpx::async( - hpx::annotated_function(gen_tile_cross_covariance, "assemble_pred"), - i, - j, - m_tile_size, - n_tile_size, - n_regressors, - sek_params, - test_input, - training_input)); + cross_covariance_tiles.push_back(detail::named_async( + "assemble_pred", i, j, m_tile_size, n_tile_size, n_regressors, sek_params, test_input, training_input)); } } - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { - prediction_tiles.push_back(hpx::async(hpx::annotated_function(gen_tile_zeros, "assemble_tiled"), m_tile_size)); + prediction_tiles.push_back(detail::named_async("assemble_tiled", m_tile_size)); } - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { - prior_K_tiles.push_back(hpx::async( - hpx::annotated_function(gen_tile_prior_covariance, "assemble_tiled"), - i, - i, - m_tile_size, - n_regressors, - sek_params, - test_input)); + prior_K_tiles.push_back(detail::named_async( + "assemble_tiled", i, i, m_tile_size, n_regressors, sek_params, test_input)); } - for (std::size_t j = 0; j < static_cast(n_tiles); j++) + for (std::size_t j = 0; j < n_tiles; j++) { - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { - t_cross_covariance_tiles.push_back(hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&gen_tile_transpose), "assemble_pred"), - m_tile_size, - n_tile_size, - cross_covariance_tiles[i * static_cast(n_tiles) + j])); + t_cross_covariance_tiles.push_back(detail::named_dataflow( + "assemble_pred", m_tile_size, n_tile_size, cross_covariance_tiles[i * n_tiles + j])); } } - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { - uncertainty_tiles.push_back( - hpx::async(hpx::annotated_function(gen_tile_zeros, "assemble_prior_inter"), m_tile_size)); + uncertainty_tiles.push_back(detail::named_async("assemble_prior_inter", m_tile_size)); } // Prediction /////////////////////////////////////////////////////////////////////////// // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(K_tiles, n_tile_size, static_cast(n_tiles)); + right_looking_cholesky_tiled(K_tiles, n_tile_size, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous triangular solve L * (L^T * alpha) = y - forward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, static_cast(n_tiles)); - backward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, static_cast(n_tiles)); + forward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); + backward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous prediction computation solve: hat(y) = cross(K) * alpha matrix_vector_tiled( - cross_covariance_tiles, - alpha_tiles, - prediction_tiles, - m_tile_size, - n_tile_size, - static_cast(n_tiles), - static_cast(m_tiles)); + cross_covariance_tiles, alpha_tiles, prediction_tiles, m_tile_size, n_tile_size, n_tiles, m_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous triangular solve L * V = cross(K)^T - forward_solve_tiled_matrix( - K_tiles, - t_cross_covariance_tiles, - n_tile_size, - m_tile_size, - static_cast(n_tiles), - static_cast(m_tiles)); + forward_solve_tiled_matrix(K_tiles, t_cross_covariance_tiles, n_tile_size, m_tile_size, n_tiles, m_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous computation diag(W) = diag(V^T * V) symmetric_matrix_matrix_diagonal_tiled( - t_cross_covariance_tiles, - uncertainty_tiles, - n_tile_size, - m_tile_size, - static_cast(n_tiles), - static_cast(m_tiles)); + t_cross_covariance_tiles, uncertainty_tiles, n_tile_size, m_tile_size, n_tiles, m_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous computation diag(Sigma) = diag(prior(K)) - diag(W) - vector_difference_tiled(prior_K_tiles, uncertainty_tiles, m_tile_size, static_cast(m_tiles)); + vector_difference_tiled(prior_K_tiles, uncertainty_tiles, m_tile_size, m_tiles); /////////////////////////////////////////////////////////////////////////// // Synchronize prediction - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { auto tile = prediction_tiles[i].get(); std::copy_n(tile.begin(), tile.size(), std::back_inserter(prediction_result)); } // Synchronize uncertainty - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { auto tile = uncertainty_tiles[i].get(); std::copy_n(tile.begin(), tile.size(), std::back_inserter(uncertainty_result)); @@ -356,11 +314,11 @@ std::vector> predict_with_full_cov( const std::vector &training_output, const std::vector &test_input, const SEKParams &sek_params, - int n_tiles, - int n_tile_size, - int m_tiles, - int m_tile_size, - int n_regressors) + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t m_tiles, + std::size_t m_tile_size, + std::size_t n_regressors) { /* * Prediction: hat(y)_M = cross(K) * K^-1 * y @@ -402,156 +360,112 @@ std::vector> predict_with_full_cov( prediction_result.reserve(test_input.size()); uncertainty_result.reserve(test_input.size()); - K_tiles.resize(static_cast(n_tiles * n_tiles)); // No reserve because of triangular structure - cross_covariance_tiles.reserve(static_cast(m_tiles) * static_cast(n_tiles)); - prediction_tiles.reserve(static_cast(m_tiles)); - alpha_tiles.reserve(static_cast(n_tiles)); + K_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure + cross_covariance_tiles.reserve(m_tiles * n_tiles); + prediction_tiles.reserve(m_tiles); + alpha_tiles.reserve(n_tiles); - t_cross_covariance_tiles.reserve(static_cast(n_tiles) * static_cast(m_tiles)); - prior_K_tiles.resize(static_cast(m_tiles * m_tiles)); - uncertainty_tiles.reserve(static_cast(m_tiles)); + t_cross_covariance_tiles.reserve(n_tiles * m_tiles); + prior_K_tiles.resize(m_tiles * m_tiles); + uncertainty_tiles.reserve(m_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous assembly - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { for (std::size_t j = 0; j <= i; j++) { - K_tiles[i * static_cast(n_tiles) + j] = hpx::async( - hpx::annotated_function(gen_tile_covariance, "assemble_tiled_K"), - i, - j, - n_tile_size, - n_regressors, - sek_params, - training_input); + K_tiles[i * n_tiles + j] = detail::named_async( + "assemble_tiled_K", i, j, n_tile_size, n_regressors, sek_params, training_input); } } - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { - alpha_tiles.push_back(hpx::async( - hpx::annotated_function(gen_tile_output, "assemble_tiled_alpha"), i, n_tile_size, training_output)); + alpha_tiles.push_back( + detail::named_async("assemble_tiled_alpha", i, n_tile_size, training_output)); } - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { - for (std::size_t j = 0; j < static_cast(n_tiles); j++) + for (std::size_t j = 0; j < n_tiles; j++) { - cross_covariance_tiles.push_back(hpx::async( - hpx::annotated_function(gen_tile_cross_covariance, "assemble_pred"), - i, - j, - m_tile_size, - n_tile_size, - n_regressors, - sek_params, - test_input, - training_input)); + cross_covariance_tiles.push_back(detail::named_async( + "assemble_pred", i, j, m_tile_size, n_tile_size, n_regressors, sek_params, test_input, training_input)); } } - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { - prediction_tiles.push_back(hpx::async(hpx::annotated_function(gen_tile_zeros, "assemble_tiled"), m_tile_size)); + prediction_tiles.push_back(detail::named_async("assemble_tiled", m_tile_size)); } // Assemble prior covariance matrix vector - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { for (std::size_t j = 0; j <= i; j++) { - prior_K_tiles[i * static_cast(m_tiles) + j] = hpx::async( - hpx::annotated_function(gen_tile_full_prior_covariance, "assemble_prior_tiled"), - i, - j, - m_tile_size, - n_regressors, - sek_params, - test_input); + prior_K_tiles[i * m_tiles + j] = detail::named_async( + "assemble_prior_tiled", i, j, m_tile_size, n_regressors, sek_params, test_input); if (i != j) { - prior_K_tiles[j * static_cast(m_tiles) + i] = hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&gen_tile_transpose), "assemble_prior_tiled"), - m_tile_size, - m_tile_size, - prior_K_tiles[i * static_cast(m_tiles) + j]); + prior_K_tiles[j * m_tiles + i] = detail::named_dataflow( + "assemble_prior_tiled", m_tile_size, m_tile_size, prior_K_tiles[i * m_tiles + j]); } } } - for (std::size_t j = 0; j < static_cast(n_tiles); j++) + for (std::size_t j = 0; j < n_tiles; j++) { - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { - t_cross_covariance_tiles.push_back(hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&gen_tile_transpose), "assemble_pred"), - m_tile_size, - n_tile_size, - cross_covariance_tiles[i * static_cast(n_tiles) + j])); + t_cross_covariance_tiles.push_back(detail::named_dataflow( + "assemble_pred", m_tile_size, n_tile_size, cross_covariance_tiles[i * n_tiles + j])); } } - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { - uncertainty_tiles.push_back(hpx::async(hpx::annotated_function(gen_tile_zeros, "assemble_tiled"), m_tile_size)); + uncertainty_tiles.push_back(detail::named_async("assemble_tiled", m_tile_size)); } // Prediction /////////////////////////////////////////////////////////////////////////// // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(K_tiles, n_tile_size, static_cast(n_tiles)); + right_looking_cholesky_tiled(K_tiles, n_tile_size, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous triangular solve L * (L^T * alpha) = y - forward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, static_cast(n_tiles)); - backward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, static_cast(n_tiles)); + forward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); + backward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous prediction computation solve: hat(y) = K_cross_cov * alpha matrix_vector_tiled( - cross_covariance_tiles, - alpha_tiles, - prediction_tiles, - m_tile_size, - n_tile_size, - static_cast(n_tiles), - static_cast(m_tiles)); + cross_covariance_tiles, alpha_tiles, prediction_tiles, m_tile_size, n_tile_size, n_tiles, m_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous triangular solve L * V = cross(K)^T - forward_solve_tiled_matrix( - K_tiles, - t_cross_covariance_tiles, - n_tile_size, - m_tile_size, - static_cast(n_tiles), - static_cast(m_tiles)); + forward_solve_tiled_matrix(K_tiles, t_cross_covariance_tiles, n_tile_size, m_tile_size, n_tiles, m_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous computation of full covariance Sigma = prior(K) - V^T * V - symmetric_matrix_matrix_tiled( - t_cross_covariance_tiles, - prior_K_tiles, - n_tile_size, - m_tile_size, - static_cast(n_tiles), - static_cast(m_tiles)); + symmetric_matrix_matrix_tiled(t_cross_covariance_tiles, prior_K_tiles, n_tile_size, m_tile_size, n_tiles, m_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous computation of uncertainty diag(Sigma) - matrix_diagonal_tiled(prior_K_tiles, uncertainty_tiles, m_tile_size, static_cast(m_tiles)); + matrix_diagonal_tiled(prior_K_tiles, uncertainty_tiles, m_tile_size, m_tiles); /////////////////////////////////////////////////////////////////////////// // Synchronize prediction - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { auto tile = prediction_tiles[i].get(); std::copy(tile.begin(), tile.end(), std::back_inserter(prediction_result)); } // Synchronize uncertainty - for (std::size_t i = 0; i < static_cast(m_tiles); i++) + for (std::size_t i = 0; i < m_tiles; i++) { auto tile = uncertainty_tiles[i].get(); std::copy(tile.begin(), tile.end(), std::back_inserter(uncertainty_result)); @@ -565,9 +479,9 @@ std::vector> predict_with_full_cov( double compute_loss(const std::vector &training_input, const std::vector &training_output, const SEKParams &sek_params, - int n_tiles, - int n_tile_size, - int n_regressors) + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors) { /* * Negative log likelihood loss: @@ -596,51 +510,44 @@ double compute_loss(const std::vector &training_input, Tiled_vector alpha_tiles; // Tiled intermediate solution // Preallocate memory - K_tiles.resize(static_cast(n_tiles * n_tiles)); // No reserve because of triangular structure - y_tiles.reserve(static_cast(n_tiles)); - alpha_tiles.reserve(static_cast(n_tiles)); + K_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure + y_tiles.reserve(n_tiles); + alpha_tiles.reserve(n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous assembly - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { for (std::size_t j = 0; j <= i; j++) { - K_tiles[i * static_cast(n_tiles) + j] = hpx::async( - hpx::annotated_function(gen_tile_covariance, "assemble_tiled_K"), - i, - j, - n_tile_size, - n_regressors, - sek_params, - training_input); + K_tiles[i * n_tiles + j] = detail::named_async( + "assemble_tiled_K", i, j, n_tile_size, n_regressors, sek_params, training_input); } } - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { - y_tiles.push_back( - hpx::async(hpx::annotated_function(gen_tile_output, "assemble_tiled_y"), i, n_tile_size, training_output)); + y_tiles.push_back(detail::named_async("assemble_tiled_y", i, n_tile_size, training_output)); } - for (std::size_t i = 0; i < static_cast(n_tiles); i++) + for (std::size_t i = 0; i < n_tiles; i++) { - alpha_tiles.push_back(hpx::async( - hpx::annotated_function(gen_tile_output, "assemble_tiled_alpha"), i, n_tile_size, training_output)); + alpha_tiles.push_back( + detail::named_async("assemble_tiled_alpha", i, n_tile_size, training_output)); } /////////////////////////////////////////////////////////////////////////// // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(K_tiles, n_tile_size, static_cast(n_tiles)); + right_looking_cholesky_tiled(K_tiles, n_tile_size, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous triangular solve L * (L^T * alpha) = y - forward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, static_cast(n_tiles)); - backward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, static_cast(n_tiles)); + forward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); + backward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); /////////////////////////////////////////////////////////////////////////// // Launch asynchronous loss computation - compute_loss_tiled(K_tiles, alpha_tiles, y_tiles, loss_value, n_tile_size, static_cast(n_tiles)); + compute_loss_tiled(K_tiles, alpha_tiles, y_tiles, loss_value, n_tile_size, n_tiles); return loss_value.get(); } @@ -717,8 +624,7 @@ optimize(const std::vector &training_input, // Launch asynchronous assembly of output y for (std::size_t i = 0; i < n_tiles; i++) { - y_tiles.push_back( - hpx::async(hpx::annotated_function(gen_tile_output, "assemble_y"), i, n_tile_size, training_output)); + y_tiles.push_back(detail::named_async("assemble_y", i, n_tile_size, training_output)); } ////////////////////////////////////////////////////////////////////////////// @@ -903,8 +809,7 @@ double optimize_step(const std::vector &training_input, // Launch asynchronous assembly of output y for (std::size_t i = 0; i < n_tiles; i++) { - y_tiles.push_back( - hpx::async(hpx::annotated_function(gen_tile_output, "assemble_y"), i, n_tile_size, training_output)); + y_tiles.push_back(detail::named_async("assemble_y", i, n_tile_size, training_output)); } ////////////////////////////////////////////////////////////////////////////// @@ -918,54 +823,31 @@ double optimize_step(const std::vector &training_input, for (std::size_t j = 0; j <= i; j++) { // Compute the distance (z_i - z_j) of K entries to reuse - hpx::shared_future> cov_dists = hpx::async( - hpx::annotated_function(gen_tile_distance, "assemble_cov_dist"), - i, - j, - n_tile_size, - n_regressors, - sek_params, - training_input); + auto cov_dists = detail::named_async( + "assemble_cov_dist", i, j, n_tile_size, n_regressors, sek_params, training_input); - K_tiles[i * n_tiles + j] = hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&gen_tile_covariance_with_distance), "assemble_K"), - i, - j, - n_tile_size, - sek_params, - cov_dists); + K_tiles[i * n_tiles + j] = detail::named_dataflow( + "assemble_K", i, j, n_tile_size, sek_params, cov_dists); if (trainable_params[0]) { - grad_l_tiles[i * n_tiles + j] = hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&gen_tile_grad_l), "assemble_gradl"), - n_tile_size, - sek_params, - cov_dists); + grad_l_tiles[i * n_tiles + j] = + detail::named_dataflow("assemble_gradl", n_tile_size, sek_params, cov_dists); if (i != j) { - grad_l_tiles[j * n_tiles + i] = hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&gen_tile_transpose), "assemble_gradl_t"), - n_tile_size, - n_tile_size, - grad_l_tiles[i * n_tiles + j]); + grad_l_tiles[j * n_tiles + i] = detail::named_dataflow( + "assemble_gradl_t", n_tile_size, n_tile_size, grad_l_tiles[i * n_tiles + j]); } } if (trainable_params[1]) { - grad_v_tiles[i * n_tiles + j] = hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&gen_tile_grad_v), "assemble_gradv"), - n_tile_size, - sek_params, - cov_dists); + grad_v_tiles[i * n_tiles + j] = + detail::named_dataflow("assemble_gradv", n_tile_size, sek_params, cov_dists); if (i != j) { - grad_v_tiles[j * n_tiles + i] = hpx::dataflow( - hpx::annotated_function(hpx::unwrapping(&gen_tile_transpose), "assemble_gradv_t"), - n_tile_size, - n_tile_size, - grad_v_tiles[i * n_tiles + j]); + grad_v_tiles[j * n_tiles + i] = detail::named_dataflow( + "assemble_gradv_t", n_tile_size, n_tile_size, grad_v_tiles[i * n_tiles + j]); } } } @@ -974,7 +856,7 @@ double optimize_step(const std::vector &training_input, // Assembly with reallocation -> optimize to only set existing values for (std::size_t i = 0; i < n_tiles; i++) { - alpha_tiles[i] = hpx::async(hpx::annotated_function(gen_tile_zeros, "assemble_tiled"), n_tile_size); + alpha_tiles[i] = detail::named_async("assemble_tiled", n_tile_size); } for (std::size_t i = 0; i < n_tiles; i++) @@ -984,12 +866,12 @@ double optimize_step(const std::vector &training_input, if (i == j) { K_inv_tiles[i * n_tiles + j] = - hpx::async(hpx::annotated_function(gen_tile_identity, "assemble_identity_matrix"), n_tile_size); + detail::named_async("assemble_identity_matrix", n_tile_size); } else { - K_inv_tiles[i * n_tiles + j] = hpx::async( - hpx::annotated_function(gen_tile_zeros, "assemble_identity_matrix"), n_tile_size * n_tile_size); + K_inv_tiles[i * n_tiles + j] = + detail::named_async("assemble_identity_matrix", n_tile_size * n_tile_size); } } } diff --git a/core/src/cpu/tiled_algorithms.cpp b/core/src/cpu/tiled_algorithms.cpp index 18b416c5..8989bb08 100644 --- a/core/src/cpu/tiled_algorithms.cpp +++ b/core/src/cpu/tiled_algorithms.cpp @@ -15,7 +15,7 @@ namespace cpu // Tiled Cholesky Algorithm -void right_looking_cholesky_tiled(Tiled_matrix &ft_tiles, int N, std::size_t n_tiles) +void right_looking_cholesky_tiled(Tiled_matrix &ft_tiles, std::size_t N, std::size_t n_tiles) { for (std::size_t k = 0; k < n_tiles; k++) { @@ -52,7 +52,7 @@ void right_looking_cholesky_tiled(Tiled_matrix &ft_tiles, int N, std::size_t n_t // Tiled Triangular Solve Algorithms -void forward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, int N, std::size_t n_tiles) +void forward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, std::size_t N, std::size_t n_tiles) { for (std::size_t k = 0; k < n_tiles; k++) { @@ -75,7 +75,7 @@ void forward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, int N, st } } -void backward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, int N, std::size_t n_tiles) +void backward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, std::size_t N, std::size_t n_tiles) { for (int k_ = static_cast(n_tiles) - 1; k_ >= 0; k_--) // int instead of std::size_t for last comparison { @@ -100,8 +100,12 @@ void backward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, int N, s } } -void forward_solve_tiled_matrix( - Tiled_matrix &ft_tiles, Tiled_matrix &ft_rhs, int N, int M, std::size_t n_tiles, std::size_t m_tiles) +void forward_solve_tiled_matrix(Tiled_matrix &ft_tiles, + Tiled_matrix &ft_rhs, + std::size_t N, + std::size_t M, + std::size_t n_tiles, + std::size_t m_tiles) { for (std::size_t c = 0; c < m_tiles; c++) { @@ -134,8 +138,12 @@ void forward_solve_tiled_matrix( } } -void backward_solve_tiled_matrix( - Tiled_matrix &ft_tiles, Tiled_matrix &ft_rhs, int N, int M, std::size_t n_tiles, std::size_t m_tiles) +void backward_solve_tiled_matrix(Tiled_matrix &ft_tiles, + Tiled_matrix &ft_rhs, + std::size_t N, + std::size_t M, + std::size_t n_tiles, + std::size_t m_tiles) { for (std::size_t c = 0; c < m_tiles; c++) { @@ -173,8 +181,8 @@ void backward_solve_tiled_matrix( void matrix_vector_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_vector, Tiled_vector &ft_rhs, - int N_row, - int N_col, + std::size_t N_row, + std::size_t N_col, std::size_t n_tiles, std::size_t m_tiles) { @@ -196,7 +204,12 @@ void matrix_vector_tiled(Tiled_matrix &ft_tiles, } void symmetric_matrix_matrix_diagonal_tiled( - Tiled_matrix &ft_tiles, Tiled_vector &ft_vector, int N, int M, std::size_t n_tiles, std::size_t m_tiles) + Tiled_matrix &ft_tiles, + Tiled_vector &ft_vector, + std::size_t N, + std::size_t M, + std::size_t n_tiles, + std::size_t m_tiles) { for (std::size_t i = 0; i < m_tiles; ++i) { @@ -209,8 +222,12 @@ void symmetric_matrix_matrix_diagonal_tiled( } } -void symmetric_matrix_matrix_tiled( - Tiled_matrix &ft_tiles, Tiled_matrix &ft_result, int N, int M, std::size_t n_tiles, std::size_t m_tiles) +void symmetric_matrix_matrix_tiled(Tiled_matrix &ft_tiles, + Tiled_matrix &ft_result, + std::size_t N, + std::size_t M, + std::size_t n_tiles, + std::size_t m_tiles) { for (std::size_t c = 0; c < m_tiles; c++) { @@ -235,7 +252,7 @@ void symmetric_matrix_matrix_tiled( } } -void vector_difference_tiled(Tiled_vector &ft_minuend, Tiled_vector &ft_subtrahend, int M, std::size_t m_tiles) +void vector_difference_tiled(Tiled_vector &ft_minuend, Tiled_vector &ft_subtrahend, std::size_t M, std::size_t m_tiles) { for (std::size_t i = 0; i < m_tiles; i++) { @@ -243,7 +260,7 @@ void vector_difference_tiled(Tiled_vector &ft_minuend, Tiled_vector &ft_subtrahe } } -void matrix_diagonal_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_vector, int M, std::size_t m_tiles) +void matrix_diagonal_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_vector, std::size_t M, std::size_t m_tiles) { for (std::size_t i = 0; i < m_tiles; i++) { @@ -255,7 +272,7 @@ void compute_loss_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_alpha, Tiled_vector &ft_y, hpx::shared_future &loss, - int N, + std::size_t N, std::size_t n_tiles) { std::vector> loss_tiled; @@ -275,7 +292,7 @@ void update_hyperparameter_tiled( const Tiled_vector &ft_alpha, const AdamParams &adam_params, SEKParams &sek_params, - int N, + std::size_t N, std::size_t n_tiles, std::size_t iter, std::size_t param_idx) diff --git a/core/src/gprat.cpp b/core/src/gprat.cpp index 2ce13252..858aa672 100644 --- a/core/src/gprat.cpp +++ b/core/src/gprat.cpp @@ -7,11 +7,9 @@ #include "gpu/gp_functions.cuh" #endif -#include - GPRAT_NS_BEGIN -GP_data::GP_data(const std::string &f_path, int n, int n_reg) : +GP_data::GP_data(const std::string &f_path, std::size_t n, std::size_t n_reg) : file_path(f_path), n_samples(n), n_regressors(n_reg) @@ -21,34 +19,34 @@ GP_data::GP_data(const std::string &f_path, int n, int n_reg) : GP::GP(std::vector input, std::vector output, - int n_tiles, - int n_tile_size, - int n_regressors, - std::vector kernel_hyperparams, + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors, + const std::vector &kernel_hyperparams, std::vector trainable_bool, std::shared_ptr target) : - training_input_(input), - training_output_(output), + training_input_(std::move(input)), + training_output_(std::move(output)), n_tiles_(n_tiles), n_tile_size_(n_tile_size), - trainable_params_(trainable_bool), - target_(target), + trainable_params_(std::move(trainable_bool)), + target_(std::move(target)), n_reg(n_regressors), kernel_params(kernel_hyperparams[0], kernel_hyperparams[1], kernel_hyperparams[2]) { } GP::GP(std::vector input, std::vector output, - int n_tiles, - int n_tile_size, - int n_regressors, - std::vector kernel_hyperparams, + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors, + const std::vector &kernel_hyperparams, std::vector trainable_bool) : - training_input_(input), - training_output_(output), + training_input_(std::move(input)), + training_output_(std::move(output)), n_tiles_(n_tiles), n_tile_size_(n_tile_size), - trainable_params_(trainable_bool), + trainable_params_(std::move(trainable_bool)), target_(std::make_shared()), n_reg(n_regressors), kernel_params(kernel_hyperparams[0], kernel_hyperparams[1], kernel_hyperparams[2]) @@ -56,18 +54,18 @@ GP::GP(std::vector input, GP::GP(std::vector input, std::vector output, - int n_tiles, - int n_tile_size, - int n_regressors, - std::vector kernel_hyperparams, + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors, + const std::vector &kernel_hyperparams, std::vector trainable_bool, int gpu_id, int n_streams) : - training_input_(input), - training_output_(output), + training_input_(std::move(input)), + training_output_(std::move(output)), n_tiles_(n_tiles), n_tile_size_(n_tile_size), - trainable_params_(trainable_bool), + trainable_params_(std::move(trainable_bool)), #if GPRAT_WITH_CUDA target_(std::make_shared(CUDA_GPU(gpu_id, n_streams))), #else @@ -102,7 +100,7 @@ std::vector GP::get_training_input() const { return training_input_; } std::vector GP::get_training_output() const { return training_output_; } -std::vector GP::predict(const std::vector &test_input, int m_tiles, int m_tile_size) +std::vector GP::predict(const std::vector &test_input, std::size_t m_tiles, std::size_t m_tile_size) { return hpx::async( [this, &test_input, m_tiles, m_tile_size]() @@ -152,7 +150,7 @@ std::vector GP::predict(const std::vector &test_input, int m_til } std::vector> -GP::predict_with_uncertainty(const std::vector &test_input, int m_tiles, int m_tile_size) +GP::predict_with_uncertainty(const std::vector &test_input, std::size_t m_tiles, std::size_t m_tile_size) { return hpx::async( [this, &test_input, m_tiles, m_tile_size]() @@ -202,7 +200,7 @@ GP::predict_with_uncertainty(const std::vector &test_input, int m_tiles, } std::vector> -GP::predict_with_full_cov(const std::vector &test_input, int m_tiles, int m_tile_size) +GP::predict_with_full_cov(const std::vector &test_input, std::size_t m_tiles, std::size_t m_tile_size) { return hpx::async( [this, &test_input, m_tiles, m_tile_size]() @@ -276,7 +274,7 @@ std::vector GP::optimize(const AdamParams &adam_params) .get(); } -double GP::optimize_step(AdamParams &adam_params, int iter) +double GP::optimize_step(AdamParams &adam_params, std::size_t iter) { return hpx::async( [this, &adam_params, iter]() diff --git a/core/src/hyperparameters.cpp b/core/src/hyperparameters.cpp index ac355e5c..2a4800ce 100644 --- a/core/src/hyperparameters.cpp +++ b/core/src/hyperparameters.cpp @@ -5,7 +5,7 @@ GPRAT_NS_BEGIN -AdamParams::AdamParams(double lr, double b1, double b2, double eps, int opt_i) : +AdamParams::AdamParams(double lr, double b1, double b2, double eps, std::size_t opt_i) : learning_rate(lr), beta1(b1), beta2(b2), diff --git a/core/src/utils.cpp b/core/src/utils.cpp index bbea471b..47935bfd 100644 --- a/core/src/utils.cpp +++ b/core/src/utils.cpp @@ -4,7 +4,7 @@ GPRAT_NS_BEGIN -int compute_train_tiles(int n_samples, int n_tile_size) +std::size_t compute_train_tiles(std::size_t n_samples, std::size_t n_tile_size) { if (n_tile_size > 0) { @@ -17,7 +17,7 @@ int compute_train_tiles(int n_samples, int n_tile_size) } } -int compute_train_tile_size(int n_samples, int n_tiles) +std::size_t compute_train_tile_size(std::size_t n_samples, std::size_t n_tiles) { if (n_tiles > 0) { @@ -30,10 +30,10 @@ int compute_train_tile_size(int n_samples, int n_tiles) } } -std::pair compute_test_tiles(int n_test, int n_tiles, int n_tile_size) +std::pair compute_test_tiles(std::size_t n_test, std::size_t n_tiles, std::size_t n_tile_size) { - int m_tiles; - int m_tile_size; + std::size_t m_tiles; + std::size_t m_tile_size; // if n_test is not divisible by (incl. smaller than) n_tile_size, use the same number of tiles if ((n_test % n_tile_size) > 0) @@ -50,10 +50,10 @@ std::pair compute_test_tiles(int n_test, int n_tiles, int n_tile_size) return { m_tiles, m_tile_size }; } -std::vector load_data(const std::string &file_path, int n_samples, int offset) +std::vector load_data(const std::string &file_path, std::size_t n_samples, std::size_t offset) { std::vector _data; - _data.resize(static_cast(n_samples + offset), 0.0); + _data.resize(n_samples + offset, 0.0); FILE *input_file = fopen(file_path.c_str(), "r"); if (input_file == NULL) @@ -62,11 +62,14 @@ std::vector load_data(const std::string &file_path, int n_samples, int o } // load data - int scanned_elements = 0; - for (int i = 0; i < n_samples; i++) + std::size_t scanned_elements = 0; + for (std::size_t i = 0; i < n_samples; i++) { - scanned_elements += - fscanf(input_file, "%lf", &_data[static_cast(i + offset)]); // scanned_elements++; + const auto r = fscanf(input_file, "%lf", &_data[(i + offset)]); + if (r > 0) + { + scanned_elements += static_cast(r); + } } fclose(input_file); diff --git a/examples/gprat_cpp/src/execute.cpp b/examples/gprat_cpp/src/execute.cpp index 97ce3d8f..fa99996f 100644 --- a/examples/gprat_cpp/src/execute.cpp +++ b/examples/gprat_cpp/src/execute.cpp @@ -15,7 +15,7 @@ int main(int argc, char *argv[]) std::size_t LOOP = 2; const std::size_t OPT_ITER = 1; - int n_test = 1024; + const std::size_t n_test = 1024; const std::size_t N_CORES = 4; const std::size_t n_tiles = 16; const std::size_t n_reg = 8; @@ -49,12 +49,12 @@ int main(int argc, char *argv[]) for (std::size_t start = START; start <= END; start = start * STEP) { - int n_train = static_cast(start); + const auto n_train = start; for (std::size_t l = 0; l < LOOP; l++) { // Compute tile sizes and number of predict tiles - int tile_size = gprat::compute_train_tile_size(n_train, n_tiles); - auto result = gprat::compute_test_tiles(n_test, n_tiles, tile_size); + const auto tile_size = gprat::compute_train_tile_size(n_train, n_tiles); + const auto result = gprat::compute_test_tiles(n_test, n_tiles, tile_size); ///////////////////// ///// hyperparams gprat::AdamParams hpar = { 0.1, 0.9, 0.999, 1e-8, OPT_ITER }; diff --git a/test/src/output_correctness.cpp b/test/src/output_correctness.cpp index 9ac5aa58..1e7ca8fc 100644 --- a/test/src/output_correctness.cpp +++ b/test/src/output_correctness.cpp @@ -104,7 +104,7 @@ gprat_results run_on_data_cpu(const std::string &train_path, const std::string & const std::size_t n_reg = 8; // Compute tile sizes and number of predict tiles - const int tile_size = gprat::compute_train_tile_size(n_train, n_tiles); + const auto tile_size = gprat::compute_train_tile_size(n_train, n_tiles); const auto test_tiles = gprat::compute_test_tiles(n_test, n_tiles, tile_size); // hyperparams @@ -146,7 +146,7 @@ gprat_results run_on_data_gpu(const std::string &train_path, const std::string & const int gpu_id = 0; const int n_streams = 1; - const int tile_size = gprat::compute_train_tile_size(n_train, n_tiles); + const auto tile_size = gprat::compute_train_tile_size(n_train, n_tiles); const auto test_tiles = gprat::compute_test_tiles(n_test, n_tiles, tile_size); gprat::GP_data training_input(train_path, n_train, n_reg); From 07998a1373212f6714baf82d5180312ed8abd045 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sat, 26 Apr 2025 18:35:20 +0200 Subject: [PATCH 16/56] feat(examples): Add command-line arguments for all algorithm parameters --- examples/gprat_cpp/src/execute.cpp | 56 +++++++++++++++++++++++------- 1 file changed, 43 insertions(+), 13 deletions(-) diff --git a/examples/gprat_cpp/src/execute.cpp b/examples/gprat_cpp/src/execute.cpp index fa99996f..4475ea8d 100644 --- a/examples/gprat_cpp/src/execute.cpp +++ b/examples/gprat_cpp/src/execute.cpp @@ -7,27 +7,57 @@ int main(int argc, char *argv[]) { + namespace po = hpx::program_options; + po::options_description desc("Allowed options"); + // clang-format off + desc.add_options() + ("help", "produce help message") + ("train_x_path", po::value()->default_value("../../../data/data_1024/training_input.txt"), "training data (x)") + ("train_y_path", po::value()->default_value("../../../data/data_1024/training_output.txt"), "training data (y)") + ("test_path", po::value()->default_value("../../../data/data_1024/test_input.txt"), "test data") + ("tiles", po::value()->default_value(16), "tiles per dimension") + ("regressors", po::value()->default_value(8), "num regressors") + ("start-cores", po::value()->default_value(2), "num CPUs to start with") + ("end-cores", po::value()->default_value(4), "num CPUs to end with") + ("start", po::value()->default_value(512), "Starting number of training samples") + ("end", po::value()->default_value(1024), "End number of training samples") + ("step", po::value()->default_value(2), "Increment of training samples") + ("loop", po::value()->default_value(2), "Number of iterations to be performed for each number of training samples") + ("opt_iter", po::value()->default_value(1), "Number of optimization iterations*/") + ; + // clang-format on + + po::variables_map vm; + po::store(po::parse_command_line(argc, argv, desc), vm); + po::notify(vm); + + if (vm.count("help")) + { + std::cout << desc << "\n"; + return 1; + } + ///////////////////// /////// configuration - std::size_t START = 512; - std::size_t END = 1024; - std::size_t STEP = 2; - std::size_t LOOP = 2; - const std::size_t OPT_ITER = 1; + std::size_t START = vm["start"].as(); + std::size_t END = vm["end"].as(); + std::size_t STEP = vm["step"].as(); + std::size_t LOOP = vm["loop"].as(); + const std::size_t OPT_ITER = vm["opt_iter"].as(); - const std::size_t n_test = 1024; - const std::size_t N_CORES = 4; - const std::size_t n_tiles = 16; - const std::size_t n_reg = 8; + const std::size_t n_test = START; + const std::size_t N_CORES = vm["end-cores"].as(); + const std::size_t n_tiles = vm["tiles"].as(); + const std::size_t n_reg = vm["regressors"].as(); - std::string train_path = "../../../data/data_1024/training_input.txt"; - std::string out_path = "../../../data/data_1024/training_output.txt"; - std::string test_path = "../../../data/data_1024/test_input.txt"; + std::string train_path = vm["train_x_path"].as(); + std::string out_path = vm["train_y_path"].as(); + std::string test_path = vm["test_path"].as(); bool use_gpu = gprat::compiled_with_cuda() && gprat::gpu_count() > 0 && argc > 1 && std::strcmp(argv[1], "--use_gpu") == 0; - for (std::size_t core = 2; core <= N_CORES; core = core * 2) + for (std::size_t core = vm["start-cores"].as(); core <= N_CORES; core = core * 2) { // Create new argc and argv to include the --hpx:threads argument std::vector args(argv, argv + argc); From 986bf4f2176d5a73423509531b3e6554587f3bd7 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Tue, 6 May 2025 23:33:06 +0200 Subject: [PATCH 17/56] fix(examples): Don't try to write results outside of the target directory --- examples/gprat_cpp/src/execute.cpp | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/examples/gprat_cpp/src/execute.cpp b/examples/gprat_cpp/src/execute.cpp index 4475ea8d..7089155e 100644 --- a/examples/gprat_cpp/src/execute.cpp +++ b/examples/gprat_cpp/src/execute.cpp @@ -31,7 +31,7 @@ int main(int argc, char *argv[]) po::store(po::parse_command_line(argc, argv, desc), vm); po::notify(vm); - if (vm.count("help")) + if (vm.contains("help")) { std::cout << desc << "\n"; return 1; @@ -205,7 +205,7 @@ int main(int argc, char *argv[]) auto total_time = end_total - start_total; // Save parameters and times to a .txt file with a header - std::ofstream outfile("../output.csv", std::ios::app); // Append mode + std::ofstream outfile("output.csv", std::ios::app); // Append mode if (outfile.tellp() == 0) { // If file is empty, write the header From 80a880ab0b2f9114dd225316ff934f77544ac534 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sat, 20 Sep 2025 22:06:46 +0200 Subject: [PATCH 18/56] feat(core): Track function invocation count as well Extends our performance counter to track #calls and runtime. --- core/include/gprat/performance_counters.hpp | 6 +++++ core/src/cpu/adapter_cblas_fp32.cpp | 28 +++++++++++++-------- core/src/cpu/adapter_cblas_fp64.cpp | 28 +++++++++++++-------- core/src/cpu/gp_algorithms.cpp | 12 ++++++++- 4 files changed, 51 insertions(+), 23 deletions(-) diff --git a/core/include/gprat/performance_counters.hpp b/core/include/gprat/performance_counters.hpp index 402cb710..e347faff 100644 --- a/core/include/gprat/performance_counters.hpp +++ b/core/include/gprat/performance_counters.hpp @@ -67,6 +67,12 @@ std::uint64_t get_and_reset_function_elapsed(bool reset) return hpx::util::get_and_reset_value(function_performance_metrics::elapsed_ns, reset); } +template +std::uint64_t get_and_reset_function_calls(bool reset) +{ + return hpx::util::get_and_reset_value(function_performance_metrics::num_calls, reset); +} + void track_tile_data_allocation(std::size_t size); void track_tile_data_deallocation(std::size_t size); diff --git a/core/src/cpu/adapter_cblas_fp32.cpp b/core/src/cpu/adapter_cblas_fp32.cpp index 29c06ec2..ca01a091 100644 --- a/core/src/cpu/adapter_cblas_fp32.cpp +++ b/core/src/cpu/adapter_cblas_fp32.cpp @@ -212,22 +212,28 @@ void register_fp32_performance_counters() // XXX: you can do this with templates, but it's quite a bit more complicated #define GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(name, fn_expr) \ hpx::performance_counters::install_counter_type( \ - name, \ + name "/time", \ get_and_reset_function_elapsed, \ #fn_expr, \ "", \ + hpx::performance_counters::counter_type::monotonically_increasing); \ + hpx::performance_counters::install_counter_type( \ + name "/calls", \ + get_and_reset_function_calls, \ + #fn_expr, \ + "", \ hpx::performance_counters::counter_type::monotonically_increasing) - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/potrf32/time", &potrf); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/trsm32/time", &trsm); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/syrk32/time", &syrk); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/gemm32/time", &gemm); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/trsv32/time", &trsv); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/gemv32/time", &gemv); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot_diag_syrk32/time", &dot_diag_syrk); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot_diag_gemm32/time", &dot_diag_gemm); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/axpy32/time", &axpy); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot32/time", &dot); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/potrf32", &potrf); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/trsm32", &trsm); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/syrk32", &syrk); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/gemm32", &gemm); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/trsv32", &trsv); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/gemv32", &gemv); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot_diag_syrk32", &dot_diag_syrk); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot_diag_gemm32", &dot_diag_gemm); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/axpy32", &axpy); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot32", &dot); #undef GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR } diff --git a/core/src/cpu/adapter_cblas_fp64.cpp b/core/src/cpu/adapter_cblas_fp64.cpp index 46cb5c3f..f2e8b927 100644 --- a/core/src/cpu/adapter_cblas_fp64.cpp +++ b/core/src/cpu/adapter_cblas_fp64.cpp @@ -213,22 +213,28 @@ void register_fp64_performance_counters() // XXX: you can do this with templates, but it's quite a bit more complicated #define GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(name, fn_expr) \ hpx::performance_counters::install_counter_type( \ - name, \ + name "/time", \ get_and_reset_function_elapsed, \ #fn_expr, \ "", \ + hpx::performance_counters::counter_type::monotonically_increasing); \ + hpx::performance_counters::install_counter_type( \ + name "/calls", \ + get_and_reset_function_calls, \ + #fn_expr, \ + "", \ hpx::performance_counters::counter_type::monotonically_increasing) - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/potrf64/time", &potrf); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/trsm64/time", &trsm); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/syrk64/time", &syrk); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/gemm64/time", &gemm); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/trsv64/time", &trsv); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/gemv64/time", &gemv); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot_diag_syrk64/time", &dot_diag_syrk); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot_diag_gemm64/time", &dot_diag_gemm); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/axpy64/time", &axpy); - GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot64/time", &dot); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/potrf64", &potrf); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/trsm64", &trsm); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/syrk64", &syrk); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/gemm64", &gemm); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/trsv64", &trsv); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/gemv64", &gemv); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot_diag_syrk64", &dot_diag_syrk); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot_diag_gemm64", &dot_diag_gemm); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/axpy64", &axpy); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/dot64", &dot); #undef GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR } diff --git a/core/src/cpu/gp_algorithms.cpp b/core/src/cpu/gp_algorithms.cpp index c99b570a..49f883dc 100644 --- a/core/src/cpu/gp_algorithms.cpp +++ b/core/src/cpu/gp_algorithms.cpp @@ -1,6 +1,6 @@ #include "gprat/cpu/gp_algorithms.hpp" - #include "gprat/tile_data.hpp" +#include "gprat/performance_counters.hpp" #include @@ -16,6 +16,7 @@ double compute_covariance_function(std::size_t n_regressors, std::span i_input, std::span j_input) { + GPRAT_TIME_FUNCTION(&compute_covariance_function); // k(z_i,z_j) = vertical_lengthscale * exp(-0.5 / lengthscale^2 * (z_i - z_j)^2) double distance = 0.0; for (std::size_t k = 0; k < n_regressors; k++) @@ -35,6 +36,7 @@ mutable_tile_data gen_tile_covariance( const SEKParams &sek_params, std::span input) { + GPRAT_TIME_FUNCTION(&gen_tile_covariance); mutable_tile_data tile(N * N); for (std::size_t i = 0; i < N; i++) { @@ -66,6 +68,7 @@ mutable_tile_data gen_tile_full_prior_covariance( const SEKParams &sek_params, std::span input) { + GPRAT_TIME_FUNCTION(&gen_tile_full_prior_covariance); mutable_tile_data tile(N * N); for (std::size_t i = 0; i < N; i++) { @@ -89,6 +92,7 @@ mutable_tile_data gen_tile_prior_covariance( const SEKParams &sek_params, std::span input) { + GPRAT_TIME_FUNCTION(&gen_tile_prior_covariance); mutable_tile_data tile(N); for (std::size_t i = 0; i < N; i++) { @@ -111,6 +115,7 @@ mutable_tile_data gen_tile_cross_covariance( std::span row_input, std::span col_input) { + GPRAT_TIME_FUNCTION(&gen_tile_cross_covariance); mutable_tile_data tile(N_row * N_col); for (std::size_t i = 0; i < N_row; i++) { @@ -131,6 +136,7 @@ mutable_tile_data gen_tile_cross_covariance( mutable_tile_data gen_tile_transpose(std::size_t N_row, std::size_t N_col, std::span tile) { + GPRAT_TIME_FUNCTION(&gen_tile_transpose); mutable_tile_data transposed(N_row * N_col); // Transpose entries for (std::size_t j = 0; j < N_col; j++) @@ -146,6 +152,7 @@ mutable_tile_data gen_tile_transpose(std::size_t N_row, std::size_t N_co mutable_tile_data gen_tile_output(std::size_t row, std::size_t N, std::span output) { + GPRAT_TIME_FUNCTION(&gen_tile_output); mutable_tile_data tile(N); std::copy(output.data() + (N * row), output.data() + (N * (row + 1)), tile.data()); return tile; @@ -153,6 +160,7 @@ mutable_tile_data gen_tile_output(std::size_t row, std::size_t N, std::s mutable_tile_data gen_tile_zeros(std::size_t N) { + GPRAT_TIME_FUNCTION(&gen_tile_zeros); mutable_tile_data tile(N); std::fill_n(tile.data(), N, 0.0); return tile; @@ -160,6 +168,7 @@ mutable_tile_data gen_tile_zeros(std::size_t N) mutable_tile_data gen_tile_identity(std::size_t N) { + GPRAT_TIME_FUNCTION(&gen_tile_identity); mutable_tile_data tile(N * N); // Initialize zero tile std::fill_n(tile.data(), N * N, 0.0); @@ -178,6 +187,7 @@ double compute_error_norm(std::size_t n_tiles, const std::vector &b, const std::vector> &tiles) { + GPRAT_TIME_FUNCTION(&compute_error_norm); double error = 0.0; for (std::size_t k = 0; k < n_tiles; k++) { From e2b0700a3a3b30fa3fb1bf8b849da477a080c752 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sat, 20 Sep 2025 22:05:15 +0200 Subject: [PATCH 19/56] refactor!(core): Add scheduler type and make algorithms use it Algorithms supporting different schedulers are templates now. Consequently, they had to be moved from .cpp to .hpp --- core/include/gprat/cpu/gp_functions.hpp | 1025 ++++++++++++++++++- core/include/gprat/cpu/tiled_algorithms.hpp | 543 +++++++++- core/include/gprat/detail/async_helpers.hpp | 41 + core/include/gprat/scheduler.hpp | 89 ++ core/src/cpu/gp_functions.cpp | 907 ---------------- core/src/cpu/tiled_algorithms.cpp | 399 +------- core/src/gprat.cpp | 351 +++---- 7 files changed, 1777 insertions(+), 1578 deletions(-) create mode 100644 core/include/gprat/scheduler.hpp diff --git a/core/include/gprat/cpu/gp_functions.hpp b/core/include/gprat/cpu/gp_functions.hpp index 1df7607b..a7aadbc1 100644 --- a/core/include/gprat/cpu/gp_functions.hpp +++ b/core/include/gprat/cpu/gp_functions.hpp @@ -3,9 +3,12 @@ #pragma once +#include "gprat/cpu/gp_algorithms.hpp" +#include "gprat/cpu/tiled_algorithms.hpp" #include "gprat/detail/config.hpp" #include "gprat/hyperparameters.hpp" #include "gprat/kernels.hpp" +#include "gprat/scheduler.hpp" #include "gprat/tile_data.hpp" #include @@ -16,10 +19,10 @@ namespace cpu { /** - * @brief Perform Cholesky decompositon (+Assebmly) + * @brief Perform Cholesky decomposition (+Assembly) * * @param training_input The training input data - * @param hyperparameters The kernel hyperparameters + * @param sek_params The kernel hyperparameters * * @param n_tiles The number of training tiles * @param n_tile_size The size of each training tile @@ -27,12 +30,55 @@ namespace cpu * * @return The tiled Cholesky factor */ +template std::vector> -cholesky(const std::vector &training_input, +cholesky(Scheduler &sched, + const std::vector &training_input, const SEKParams &sek_params, std::size_t n_tiles, std::size_t n_tile_size, - std::size_t n_regressors); + std::size_t n_regressors) +{ + // Tiled covariance matrix K_NxN + auto K_tiles = make_tiled_dataset( + sched, + n_tiles * n_tiles, + [&](std::size_t tile_index) + { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + + for (std::size_t row = 0; row < n_tiles; row++) + { + for (std::size_t col = 0; col <= row; col++) + { + K_tiles[row * n_tiles + col] = detail::named_make_tile( + sched, + schedule::covariance_tile(sched, n_tiles, row, col), + "assemble_tiled_K", + K_tiles[row * n_tiles + col], + row, + col, + n_tile_size, + n_regressors, + sek_params, + training_input); + } + } + + // Launch asynchronous Cholesky decomposition: K = L * L^T + right_looking_cholesky_tiled(sched, K_tiles, n_tile_size, n_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Synchronize + std::vector> result(n_tiles * n_tiles); + for (std::size_t i = 0; i < n_tiles; i++) + { + for (std::size_t j = 0; j <= i; j++) + { + result[i * n_tiles + j] = K_tiles[i * n_tiles + j].get(); + } + } + return result; +} /** * @brief Compute the predictions without uncertainties. @@ -49,8 +95,10 @@ cholesky(const std::vector &training_input, * * @return A vector containing the predictions */ +template std::vector -predict(const std::vector &training_input, +predict(Scheduler &sched, + const std::vector &training_input, const std::vector &training_output, const std::vector &test_input, const SEKParams &sek_params, @@ -58,7 +106,129 @@ predict(const std::vector &training_input, std::size_t n_tile_size, std::size_t m_tiles, std::size_t m_tile_size, - std::size_t n_regressors); + std::size_t n_regressors) +{ + /* + * Prediction: hat(y)_M = cross(K)_MxN * K^-1_NxN * y_N + * - Covariance matrix K_NxN + * - Cross-covariance cross(K)_MxN + * - Training output y_N + * - Prediction output hat(y)_M + * + * Algorithm: + * 1: Compute lower triangular part of covariance matrix K + * 2: Compute Cholesky factor L of K + * 3: Compute prediction hat(y): + * - triangular solve L * beta = y + * - triangular solve L^T * alpha = beta + * - compute hat(y) = cross(K) * alpha + */ + + /////////////////////////////////////////////////////////////////////////// + // Cholesky + + // Tiled covariance matrix K_NxN + auto K_tiles = make_tiled_dataset( + sched, + n_tiles * n_tiles, + [&](std::size_t tile_index) + { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + + for (std::size_t row = 0; row < n_tiles; row++) + { + for (std::size_t col = 0; col <= row; col++) + { + K_tiles[row * n_tiles + col] = detail::named_make_tile( + sched, + schedule::covariance_tile(sched, n_tiles, row, col), + "assemble_tiled_K", + K_tiles[row * n_tiles + col], + row, + col, + n_tile_size, + n_regressors, + sek_params, + training_input); + } + } + + // Launch asynchronous Cholesky decomposition: K = L * L^T + right_looking_cholesky_tiled(sched, K_tiles, n_tile_size, n_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Prediction + + // Tiled cross_covariance matrix K_NxM + auto cross_covariance_tiles = make_tiled_dataset( + sched, + m_tiles * n_tiles, + [&](std::size_t tile_index) + { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + // Tiled solution + auto prediction_tiles = make_tiled_dataset( + sched, m_tiles, [&](std::size_t tile_index) { return schedule::prediction_tile(sched, m_tiles, tile_index); }); + // Tiled intermediate solution + auto alpha_tiles = make_tiled_dataset( + sched, n_tiles, [&](std::size_t tile_index) { return schedule::alpha_tile(sched, n_tiles, tile_index); }); + + for (std::size_t i = 0; i < n_tiles; i++) + { + alpha_tiles[i] = detail::named_make_tile( + sched, + schedule::alpha_tile(sched, n_tiles, i), + "assemble_tiled_alpha", + alpha_tiles[i], + i, + n_tile_size, + training_output); + } + + for (std::size_t i = 0; i < m_tiles; i++) + { + for (std::size_t j = 0; j < n_tiles; j++) + { + cross_covariance_tiles[i * n_tiles + j] = detail::named_make_tile( + sched, + schedule::cross_covariance_tile(sched, n_tiles, i, j), + "assemble_pred", + cross_covariance_tiles[i * n_tiles + j], + i, + j, + m_tile_size, + n_tile_size, + n_regressors, + sek_params, + test_input, + training_input); + } + } + + for (std::size_t i = 0; i < m_tiles; i++) + { + prediction_tiles[i] = detail::named_make_tile( + sched, schedule::prediction_tile(sched, m_tiles, i), "assemble_tiled", prediction_tiles[i], m_tile_size); + } + + // Launch asynchronous triangular solve L * (L^T * alpha) = y + forward_solve_tiled(sched, K_tiles, alpha_tiles, n_tile_size, n_tiles); + backward_solve_tiled(sched, K_tiles, alpha_tiles, n_tile_size, n_tiles); + + // Launch asynchronous prediction computation solve: \hat{y} = K_cross_cov * alpha + matrix_vector_tiled( + sched, cross_covariance_tiles, alpha_tiles, prediction_tiles, m_tile_size, n_tile_size, n_tiles, m_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Synchronize prediction + // Preallocate memory + std::vector prediction_result; + prediction_result.reserve(test_input.size()); + for (std::size_t i = 0; i < m_tiles; i++) + { + mutable_tile_data tile = prediction_tiles[i].get(); + std::copy_n(tile.data(), tile.size(), std::back_inserter(prediction_result)); + } + return prediction_result; +} /** * @brief Compute the predictions with uncertainties. @@ -75,7 +245,9 @@ predict(const std::vector &training_input, * * @return A vector containing the prediction vector and the uncertainty vector */ +template std::vector> predict_with_uncertainty( + Scheduler &sched, const std::vector &training_input, const std::vector &training_output, const std::vector &test_input, @@ -84,7 +256,211 @@ std::vector> predict_with_uncertainty( std::size_t n_tile_size, std::size_t m_tiles, std::size_t m_tile_size, - std::size_t n_regressors); + std::size_t n_regressors) +{ + /* + * Prediction: hat(y) = cross(K) * K^-1 * y + * Uncertainty: diag(Sigma) = diag(prior(K)) * diag(cross(K)^T * K^-1 * cross(K)) + * - Covariance matrix K_NxN + * - Cross-covariance cross(K)_MxN + * - Prior covariance prior(K)_MxM + * - Training output y_N + * - Prediction output hat(y)_M + * - Posterior covariance matrix Sigma_MxM + * + * Algorithm: + * 1: Compute lower triangular part of covariance matrix K + * 2: Compute Cholesky factor L of K + * 3: Compute prediction hat(y): + * - triangular solve L * beta = y + * - triangular solve L^T * alpha = beta + * - compute hat(y) = cross(K) * alpha + * 4: Compute uncertainty diag(Sigma): + * - triangular solve L * V = cross(K)^T + * - compute diag(W) = diag(V^T * V) + * - compute diag(Sigma) = diag(prior(K)) - diag(W) + */ + + /////////////////////////////////////////////////////////////////////////// + // Cholesky + + // Tiled covariance matrix K_NxN + auto K_tiles = make_tiled_dataset( + sched, + n_tiles * n_tiles, + [&](std::size_t tile_index) + { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + + for (std::size_t row = 0; row < n_tiles; row++) + { + for (std::size_t col = 0; col <= row; col++) + { + K_tiles[row * n_tiles + col] = detail::named_make_tile( + sched, + schedule::covariance_tile(sched, n_tiles, row, col), + "assemble_tiled_K", + K_tiles[row * n_tiles + col], + row, + col, + n_tile_size, + n_regressors, + sek_params, + training_input); + } + } + + // Launch asynchronous Cholesky decomposition: K = L * L^T + right_looking_cholesky_tiled(sched, K_tiles, n_tile_size, n_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Prediction + + // Tiled intermediate solution + auto alpha_tiles = make_tiled_dataset( + sched, n_tiles, [&](std::size_t tile_index) { return schedule::alpha_tile(sched, n_tiles, tile_index); }); + for (std::size_t i = 0; i < n_tiles; i++) + { + alpha_tiles[i] = detail::named_make_tile( + sched, + schedule::alpha_tile(sched, n_tiles, i), + "assemble_tiled_alpha", + alpha_tiles[i], + i, + n_tile_size, + training_output); + } + + // Tiled cross_covariance matrix K_NxM + auto cross_covariance_tiles = make_tiled_dataset( + sched, + m_tiles * n_tiles, + [&](std::size_t tile_index) + { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + for (std::size_t i = 0; i < m_tiles; i++) + { + for (std::size_t j = 0; j < n_tiles; j++) + { + cross_covariance_tiles[i * n_tiles + j] = detail::named_make_tile( + sched, + schedule::cross_covariance_tile(sched, n_tiles, i, j), + "assemble_pred", + cross_covariance_tiles[i * n_tiles + j], + i, + j, + m_tile_size, + n_tile_size, + n_regressors, + sek_params, + test_input, + training_input); + } + } + + // Tiled solution + auto prediction_tiles = make_tiled_dataset( + sched, m_tiles, [&](std::size_t tile_index) { return schedule::prediction_tile(sched, m_tiles, tile_index); }); + for (std::size_t i = 0; i < m_tiles; i++) + { + prediction_tiles[i] = detail::named_make_tile( + sched, schedule::prediction_tile(sched, m_tiles, i), "assemble_tiled", prediction_tiles[i], m_tile_size); + } + + // Launch asynchronous triangular solve L * (L^T * alpha) = y + forward_solve_tiled(sched, K_tiles, alpha_tiles, n_tile_size, n_tiles); + backward_solve_tiled(sched, K_tiles, alpha_tiles, n_tile_size, n_tiles); + + // Launch asynchronous prediction computation solve: \hat{y} = K_cross_cov * alpha + matrix_vector_tiled( + sched, cross_covariance_tiles, alpha_tiles, prediction_tiles, m_tile_size, n_tile_size, n_tiles, m_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Uncertainty + + // Tiled transposed cross_covariance matrix K_MxN + auto t_cross_covariance_tiles = make_tiled_dataset( + sched, + n_tiles * m_tiles, + [&](std::size_t tile_index) + { return schedule::t_cross_covariance_tile(sched, m_tiles, tile_index / m_tiles, tile_index % m_tiles); }); + for (std::size_t j = 0; j < n_tiles; j++) + { + for (std::size_t i = 0; i < m_tiles; i++) + { + t_cross_covariance_tiles[j * m_tiles + i] = detail::named_make_tile( + sched, + schedule::t_cross_covariance_tile(sched, m_tiles, j, i), + "assemble_pred", + t_cross_covariance_tiles[j * m_tiles + i], + m_tile_size, + n_tile_size, + cross_covariance_tiles[i * n_tiles + j]); + } + } + + // Tiled prior covariance matrix diagonal diag(K_MxM) + auto prior_K_tiles = make_tiled_dataset( + sched, m_tiles, [&](std::size_t tile_index) { return schedule::prior_K_tile(sched, n_tiles, 0, tile_index); }); + for (std::size_t i = 0; i < m_tiles; i++) + { + prior_K_tiles[i] = detail::named_make_tile( + sched, + schedule::prior_K_tile(sched, m_tiles, 0, i), + "assemble_tiled", + prior_K_tiles[i], + i, + i, + m_tile_size, + n_regressors, + sek_params, + test_input); + } + + // Tiled uncertainty solution + auto uncertainty_tiles = make_tiled_dataset( + sched, m_tiles, [&](std::size_t tile_index) { return schedule::uncertainty_tile(sched, m_tiles, tile_index); }); + for (std::size_t i = 0; i < m_tiles; i++) + { + uncertainty_tiles[i] = detail::named_make_tile( + sched, + schedule::uncertainty_tile(sched, m_tiles, i), + "assemble_prior_inter", + uncertainty_tiles[i], + m_tile_size); + } + + // Launch asynchronous triangular solve L * V = cross(K)^T + forward_solve_tiled_matrix(sched, K_tiles, t_cross_covariance_tiles, n_tile_size, m_tile_size, n_tiles, m_tiles); + + // Launch asynchronous computation diag(W) = diag(V^T * V) + symmetric_matrix_matrix_diagonal_tiled( + sched, t_cross_covariance_tiles, uncertainty_tiles, n_tile_size, m_tile_size, n_tiles, m_tiles); + + // Launch asynchronous computation diag(Sigma) = diag(prior(K)) - diag(W) + vector_difference_tiled(sched, prior_K_tiles, uncertainty_tiles, m_tile_size, m_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Preallocate memory + std::vector prediction_result; + std::vector uncertainty_result; + prediction_result.reserve(test_input.size()); + uncertainty_result.reserve(test_input.size()); + + // Synchronize prediction + for (std::size_t i = 0; i < m_tiles; i++) + { + mutable_tile_data tile = prediction_tiles[i].get(); + std::copy_n(tile.begin(), tile.size(), std::back_inserter(prediction_result)); + } + + // Synchronize uncertainty + for (std::size_t i = 0; i < m_tiles; i++) + { + mutable_tile_data tile = uncertainty_tiles[i].get(); + std::copy_n(tile.begin(), tile.size(), std::back_inserter(uncertainty_result)); + } + + return std::vector>{ std::move(prediction_result), std::move(uncertainty_result) }; +} /** * @brief Compute the predictions with full covariance matrix. @@ -92,7 +468,7 @@ std::vector> predict_with_uncertainty( * @param training_input The training input data * @param training_output The raining output data * @param test_input The test input data - * @param hyperparameters The kernel hyperparameters + * @param sek_params The kernel hyperparameters * @param n_tiles The number of training tiles * @param n_tile_size The size of each training tile * @param m_tiles The number of test tiles @@ -101,35 +477,350 @@ std::vector> predict_with_uncertainty( * * @return A vector containing the prediction vector and the full posterior covariance matrix */ +template std::vector> predict_with_full_cov( + Scheduler &sched, const std::vector &training_input, const std::vector &training_output, - const std::vector &test_data, + const std::vector &test_input, const SEKParams &sek_params, std::size_t n_tiles, std::size_t n_tile_size, std::size_t m_tiles, std::size_t m_tile_size, - std::size_t n_regressors); + std::size_t n_regressors) +{ + /* + * Prediction: hat(y)_M = cross(K) * K^-1 * y + * Full covariance: Sigma = prior(K) - cross(K)^T * K^-1 * cross(K) + * - Covariance matrix K_NxN + * - Cross-covariance cross(K)_MxN + * - Prior covariance prior(K)_MxM + * - Training output y_N + * - Prediction output hat(y)_M + * - Posterior covariance matrix Sigma_MxM + * + * Algorithm: + * 1: Compute lower triangular part of covariance matrix K + * 2: Compute Cholesky factor L of K + * 3: Compute prediction hat(y): + * - triangular solve L * beta = y + * - triangular solve L^T * alpha = beta + * - compute hat(y) = cross(K) * alpha + * 4: Compute full covariance matrix Sigma: + * - triangular solve L * V = cross(K)^T + * - compute W = V^T * V + * - compute Sigma = prior(K) - W + * 5: Compute diag(Sigma) + */ + + /////////////////////////////////////////////////////////////////////////// + // Cholesky + + // Tiled covariance matrix K_NxN + auto K_tiles = make_tiled_dataset( + sched, + n_tiles * n_tiles, + [&](std::size_t tile_index) + { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + for (std::size_t row = 0; row < n_tiles; row++) + { + for (std::size_t col = 0; col <= row; col++) + { + K_tiles[row * n_tiles + col] = detail::named_make_tile( + sched, + schedule::covariance_tile(sched, n_tiles, row, col), + "assemble_tiled_K", + K_tiles[row * n_tiles + col], + row, + col, + n_tile_size, + n_regressors, + sek_params, + training_input); + } + } + + // Launch asynchronous Cholesky decomposition: K = L * L^T + right_looking_cholesky_tiled(sched, K_tiles, n_tile_size, n_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Prediction + + // Tiled intermediate solution + auto alpha_tiles = make_tiled_dataset( + sched, n_tiles, [&](std::size_t tile_index) { return schedule::alpha_tile(sched, n_tiles, tile_index); }); + for (std::size_t i = 0; i < n_tiles; i++) + { + alpha_tiles[i] = detail::named_make_tile( + sched, + schedule::alpha_tile(sched, n_tiles, i), + "assemble_tiled_alpha", + alpha_tiles[i], + i, + n_tile_size, + training_output); + } + + // Tiled cross_covariance matrix K_NxM + auto cross_covariance_tiles = make_tiled_dataset( + sched, + m_tiles * n_tiles, + [&](std::size_t tile_index) + { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + for (std::size_t i = 0; i < m_tiles; i++) + { + for (std::size_t j = 0; j < n_tiles; j++) + { + cross_covariance_tiles[i * n_tiles + j] = detail::named_make_tile( + sched, + schedule::cross_covariance_tile(sched, n_tiles, i, j), + "assemble_pred", + cross_covariance_tiles[i * n_tiles + j], + i, + j, + m_tile_size, + n_tile_size, + n_regressors, + sek_params, + test_input, + training_input); + } + } + + // Tiled solution + auto prediction_tiles = make_tiled_dataset( + sched, m_tiles, [&](std::size_t tile_index) { return schedule::prediction_tile(sched, n_tiles, tile_index); }); + for (std::size_t i = 0; i < m_tiles; i++) + { + prediction_tiles[i] = detail::named_make_tile( + sched, schedule::prediction_tile(sched, m_tiles, i), "assemble_tiled", prediction_tiles[i], m_tile_size); + } + + // Launch asynchronous triangular solve L * (L^T * alpha) = y + forward_solve_tiled(sched, K_tiles, alpha_tiles, n_tile_size, n_tiles); + backward_solve_tiled(sched, K_tiles, alpha_tiles, n_tile_size, n_tiles); + + // Launch asynchronous prediction computation solve: \hat{y} = K_cross_cov * alpha + matrix_vector_tiled( + sched, cross_covariance_tiles, alpha_tiles, prediction_tiles, m_tile_size, n_tile_size, n_tiles, m_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Uncertainty + + // Tiled transposed cross_covariance matrix K_MxN + auto t_cross_covariance_tiles = make_tiled_dataset( + sched, + n_tiles * m_tiles, + [&](std::size_t tile_index) + { return schedule::t_cross_covariance_tile(sched, m_tiles, tile_index / m_tiles, tile_index % m_tiles); }); + for (std::size_t j = 0; j < n_tiles; j++) + { + for (std::size_t i = 0; i < m_tiles; i++) + { + t_cross_covariance_tiles[j * m_tiles + i] = detail::named_make_tile( + sched, + schedule::t_cross_covariance_tile(sched, m_tiles, j, i), + "assemble_pred", + t_cross_covariance_tiles[j * m_tiles + i], + m_tile_size, + n_tile_size, + cross_covariance_tiles[i * n_tiles + j]); + } + } + + // Tiled prior covariance matrix K_MxM + auto prior_K_tiles = make_tiled_dataset( + sched, + m_tiles * m_tiles, + [&](std::size_t tile_index) + { return schedule::prior_K_tile(sched, n_tiles, tile_index / m_tiles, tile_index % m_tiles); }); + for (std::size_t i = 0; i < m_tiles; i++) + { + for (std::size_t j = 0; j <= i; j++) + { + prior_K_tiles[i * m_tiles + j] = detail::named_make_tile( + sched, + schedule::prior_K_tile(sched, m_tiles, i, j), + "assemble_prior_tiled", + prior_K_tiles[i * m_tiles + j], + i, + j, + m_tile_size, + n_regressors, + sek_params, + test_input); + + if (i != j) + { + prior_K_tiles[j * m_tiles + i] = detail::named_make_tile( + sched, + schedule::prior_K_tile(sched, m_tiles, j, i), + "assemble_prior_tiled", + prior_K_tiles[j * m_tiles + i], + m_tile_size, + m_tile_size, + prior_K_tiles[i * m_tiles + j]); + } + } + } + + // Tiled uncertainty solution + auto uncertainty_tiles = make_tiled_dataset( + sched, m_tiles, [&](std::size_t tile_index) { return schedule::uncertainty_tile(sched, m_tiles, tile_index); }); + for (std::size_t i = 0; i < m_tiles; i++) + { + uncertainty_tiles[i] = detail::named_make_tile( + sched, + schedule::uncertainty_tile(sched, m_tiles, i), + "assemble_prior_inter", + uncertainty_tiles[i], + m_tile_size); + } + + // Launch asynchronous triangular solve L * V = cross(K)^T + forward_solve_tiled_matrix(sched, K_tiles, t_cross_covariance_tiles, n_tile_size, m_tile_size, n_tiles, m_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Launch asynchronous computation of full covariance Sigma = prior(K) - V^T * V + symmetric_matrix_matrix_tiled( + sched, t_cross_covariance_tiles, prior_K_tiles, n_tile_size, m_tile_size, n_tiles, m_tiles); + /////////////////////////////////////////////////////////////////////////// + // Launch asynchronous computation of uncertainty diag(Sigma) + matrix_diagonal_tiled(sched, prior_K_tiles, uncertainty_tiles, m_tile_size, m_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Preallocate memory + std::vector prediction_result; + std::vector uncertainty_result; + prediction_result.reserve(test_input.size()); + uncertainty_result.reserve(test_input.size()); + + // Synchronize prediction + for (std::size_t i = 0; i < m_tiles; i++) + { + mutable_tile_data tile = prediction_tiles[i].get(); + std::copy_n(tile.begin(), tile.size(), std::back_inserter(prediction_result)); + } + + // Synchronize uncertainty + for (std::size_t i = 0; i < m_tiles; i++) + { + mutable_tile_data tile = uncertainty_tiles[i].get(); + std::copy_n(tile.begin(), tile.size(), std::back_inserter(uncertainty_result)); + } + + return std::vector>{ std::move(prediction_result), std::move(uncertainty_result) }; +} + +/////////////////////////////////////////////////////////////////////////// +// OPTIMIZATION /** * @brief Compute loss for given data and Gaussian process model * * @param training_input The training input data * @param training_output The raining output data - * @param hyperparameters The kernel hyperparameters + * @param sek_params The kernel hyperparameters * @param n_tiles The number of training tiles * @param n_tile_size The size of each training tile * @param n_regressors The number of regressors * * @return The loss */ -double compute_loss(const std::vector &training_input, +template +double calculate_loss(Scheduler &sched, + const std::vector &training_input, const std::vector &training_output, const SEKParams &sek_params, std::size_t n_tiles, std::size_t n_tile_size, - std::size_t n_regressors); + std::size_t n_regressors) +{ + /* + * Negative log likelihood loss: + * loss(theta) = 0.5 * ( log(det(K)) - y^T * K^-1 * y - N * log(2 * pi) ) + * - Covariance matrix K(theta)_NxN + * - Training output y_N + * - Hyperparameters theta ={ v, l, v_n } + * + * Algorithm: + * 1: Compute lower triangular part of covariance matrix K + * 2: Compute Cholesky factor L of K + * 3: Compute prediction alpha = K^-1 * y: + * - triangular solve L * beta = y + * - triangular solve L^T * alpha = beta + * 5: Compute beta = K^-1 * y + * 6: Compute negative log likelihood loss + * - Calculate sum_i^N log(L_ii^2) + * - Calculate y^T * beta + * - Add constant N * log (2 * pi) + */ + + // Tiled covariance matrix K_NxN + auto K_tiles = make_tiled_dataset( + sched, + n_tiles * n_tiles, + [&](std::size_t tile_index) + { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + for (std::size_t row = 0; row < n_tiles; row++) + { + for (std::size_t col = 0; col <= row; col++) + { + K_tiles[row * n_tiles + col] = detail::named_make_tile( + sched, + schedule::covariance_tile(sched, n_tiles, row, col), + "assemble_tiled_K", + K_tiles[row * n_tiles + col], + row, + col, + n_tile_size, + n_regressors, + sek_params, + training_input); + } + } + + // Tiled intermediate solution + auto alpha_tiles = make_tiled_dataset( + sched, n_tiles, [&](std::size_t tile_index) { return schedule::alpha_tile(sched, n_tiles, tile_index); }); + for (std::size_t i = 0; i < n_tiles; i++) + { + alpha_tiles[i] = detail::named_make_tile( + sched, + schedule::alpha_tile(sched, n_tiles, i), + "assemble_tiled_alpha", + alpha_tiles[i], + i, + n_tile_size, + training_output); + } + + // Tiled output + auto y_tiles = make_tiled_dataset( + sched, n_tiles, [&](std::size_t tile_index) { return schedule::prediction_tile(sched, n_tiles, tile_index); }); + for (std::size_t i = 0; i < n_tiles; i++) + { + y_tiles[i] = detail::named_make_tile( + sched, + schedule::prediction_tile(sched, n_tiles, i), + "assemble_tiled_alpha", + y_tiles[i], + i, + n_tile_size, + training_output); + } + + // Launch asynchronous Cholesky decomposition: K = L * L^T + right_looking_cholesky_tiled(sched, K_tiles, n_tile_size, n_tiles); + + // Launch asynchronous triangular solve L * (L^T * alpha) = y + forward_solve_tiled(sched, K_tiles, alpha_tiles, n_tile_size, n_tiles); + backward_solve_tiled(sched, K_tiles, alpha_tiles, n_tile_size, n_tiles); + + // Launch asynchronous loss computation + return compute_loss_tiled(sched, K_tiles, alpha_tiles, y_tiles, n_tile_size, n_tiles).get(); +} /** * @brief Perform optimization for a given number of iterations @@ -141,21 +832,289 @@ double compute_loss(const std::vector &training_input, * @param n_tile_size The size of each training tile * @param n_regressors The number of regressors * - * @param hyperparams The Adam optimizer hyperparameters - * @param hyperparameters The kernel hyperparameters - * @param trainable_params The vector containing a bool wheather to train a hyperparameter + * @param adam_params The Adam optimizer hyperparameters + * @param sek_params The kernel hyperparameters + * @param trainable_params The vector containing a bool whether to train a hyperparameter * * @return A vector containing the loss values of each iteration */ +template std::vector -optimize(const std::vector &training_input, +optimize(Scheduler &sched, + const std::vector &training_input, const std::vector &training_output, std::size_t n_tiles, std::size_t n_tile_size, std::size_t n_regressors, const AdamParams &adam_params, SEKParams &sek_params, - std::vector trainable_params); + std::vector trainable_params, + std::size_t start_iter = 0) +{ + /* + * - Hyperparameters theta={v, l, v_n} + * - Covariance matrix K(theta) + * - Training ouput y + * + * Algorithm: + * for opt_iter: + * 1: Compute distance for entries of covariance matrix K + * 2: Compute lower triangular part of K with distance + * 3: Compute lower triangular gradients for delta(K)/delta(v), and delta(K)/delta(l) with distance + * + * 4: Compute Cholesky factor L of K + * 5: Compute K^-1: + * - triangular solve L * {} = I + * - triangular solve L^T * K^-1 = {} + * 6: Compute beta = K^-1 * y + * + * 7: Compute negative log likelihood loss + * - Calculate 0.5 sum_i^N log(L_ii^2) + * - Calculate 0.5 y^T * beta + * - Add constant N / 2 * log (2 * pi) + * + * 8: Compute delta(loss)/delta(param_i) + * - Compute trace(K^-1 * delta(K)/delta(theta_i)) + * - Compute beta^T * delta(K)/delta(theta_i) * beta + * 9: Update hyperparameters theta with Adam optimizer + * - m_T = beta1 * m_T-1 + (1 - beta1) * g_T + * - w_T = beta2 + w_T-1 + (1 - beta2) * g_T^2 + * - nu_T = nu * sqrt(1 - beta2_T) / (1 - beta1_T) + * - theta_T = theta_T-1 - nu_T * m_T / (sqrt(w_T) + epsilon) + * endfor + */ + + // data holder for computed loss values + std::vector losses; + losses.reserve(static_cast(adam_params.opt_iter)); + + // Tiled output + auto y_tiles = make_tiled_dataset( + sched, n_tiles, [&](std::size_t tile_index) { return schedule::prediction_tile(sched, n_tiles, tile_index); }); + // Launch asynchronous assembly of output y + for (std::size_t i = 0; i < n_tiles; i++) + { + y_tiles[i] = detail::named_make_tile( + sched, + schedule::prediction_tile(sched, n_tiles, i), + "assemble_y", + y_tiles[i], + i, + n_tile_size, + training_output); + } + + ////////////////////////////////////////////////////////////////////////////// + // per-loop tiles + + // Tiled covariance matrix K_NxN + auto K_tiles = make_tiled_dataset( + sched, + n_tiles * n_tiles, + [&](std::size_t tile_index) + { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + + // Tiled inverse covariance matrix K^-1_NxN + auto K_inv_tiles = make_tiled_dataset( + sched, + n_tiles * n_tiles, + [&](std::size_t tile_index) + { return schedule::K_inv_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + + // Tiled intermediate solution + auto alpha_tiles = make_tiled_dataset( + sched, n_tiles, [&](std::size_t tile_index) { return schedule::alpha_tile(sched, n_tiles, tile_index); }); + + // Tiled future data structures for gradients + + // Tiled covariance with gradient v + auto grad_v_tiles = make_tiled_dataset( + sched, + n_tiles * n_tiles, + [&](std::size_t tile_index) + { return schedule::K_grad_v_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + + // Tiled covariance with gradient l + auto grad_l_tiles = make_tiled_dataset( + sched, + n_tiles * n_tiles, + [&](std::size_t tile_index) + { return schedule::K_grad_l_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + + auto inter_alpha = make_tiled_dataset( + sched, n_tiles, [&](std::size_t tile_index) { return schedule::inter_alpha_tile(sched, n_tiles, tile_index); }); + + auto diag_tiles = make_tiled_dataset( + sched, n_tiles, [&](std::size_t tile_index) { return schedule::diag_tile(sched, n_tiles, tile_index); }); + + ////////////////////////////////////////////////////////////////////////////// + // Perform optimization + for (std::size_t iter = start_iter; iter < static_cast(adam_params.opt_iter); iter++) + { + /////////////////////////////////////////////////////////////////////////// + // Launch asynchronous assembly of tiled covariance matrix, derivative of covariance matrix + // vector w.r.t. to vertical lengthscale and derivative of covariance + // matrix vector w.r.t. to lengthscale + for (std::size_t i = 0; i < n_tiles; i++) + { + for (std::size_t j = 0; j <= i; j++) + { + // Compute the distance (z_i - z_j) of K entries to reuse + hpx::shared_future> cov_dists = detail::named_async( + "assemble_cov_dist", i, j, n_tile_size, n_regressors, sek_params, training_input); + + K_tiles[i * n_tiles + j] = detail::named_make_tile( + sched, + schedule::covariance_tile(sched, n_tiles, i, j), + "assemble_K", + K_tiles[i * n_tiles + j], + i, + j, + n_tile_size, + sek_params, + cov_dists); + if (trainable_params[0]) + { + grad_l_tiles[i * n_tiles + j] = detail::named_make_tile( + sched, + schedule::K_grad_l_tile(sched, n_tiles, i, j), + "assemble_gradl", + grad_l_tiles[i * n_tiles + j], + n_tile_size, + sek_params, + cov_dists); + if (i != j) + { + grad_l_tiles[j * n_tiles + i] = detail::named_make_tile( + sched, + schedule::K_grad_l_tile(sched, n_tiles, j, i), + "assemble_gradl_t", + grad_l_tiles[j * n_tiles + i], + n_tile_size, + n_tile_size, + grad_l_tiles[i * n_tiles + j]); + } + } + + if (trainable_params[1]) + { + grad_v_tiles[i * n_tiles + j] = detail::named_make_tile( + sched, + schedule::K_grad_v_tile(sched, n_tiles, i, j), + "assemble_gradv", + grad_v_tiles[i * n_tiles + j], + n_tile_size, + sek_params, + cov_dists); + if (i != j) + { + grad_v_tiles[j * n_tiles + i] = detail::named_make_tile( + sched, + schedule::K_grad_v_tile(sched, n_tiles, j, i), + "assemble_gradv_t", + grad_v_tiles[j * n_tiles + i], + n_tile_size, + n_tile_size, + grad_v_tiles[i * n_tiles + j]); + } + } + } + } + + // Assembly with reallocation -> optimize to only set existing values + for (std::size_t i = 0; i < n_tiles; i++) + { + alpha_tiles[i] = detail::named_make_tile( + sched, schedule::alpha_tile(sched, n_tiles, i), "assemble_tiled_alpha", alpha_tiles[i], n_tile_size); + } + + for (std::size_t i = 0; i < n_tiles; i++) + { + for (std::size_t j = 0; j < n_tiles; j++) + { + if (i == j) + { + K_inv_tiles[i * n_tiles + j] = detail::named_make_tile( + sched, + schedule::K_inv_tile(sched, n_tiles, i, j), + "assemble_identity_matrix", + K_inv_tiles[i * n_tiles + j], + n_tile_size); + } + else + { + K_inv_tiles[i * n_tiles + j] = detail::named_make_tile( + sched, + schedule::K_inv_tile(sched, n_tiles, i, j), + "assemble_identity_matrix", + K_inv_tiles[i * n_tiles + j], + n_tile_size * n_tile_size); + } + } + } + + /////////////////////////////////////////////////////////////////////////// + // Launch asynchronous Cholesky decomposition: K = L * L^T + right_looking_cholesky_tiled(sched, K_tiles, n_tile_size, n_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Launch asynchronous compute K^-1 through L* (L^T * X) = I + forward_solve_tiled_matrix(sched, K_tiles, K_inv_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); + backward_solve_tiled_matrix(sched, K_tiles, K_inv_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Launch asynchronous compute beta = inv(K) * y + matrix_vector_tiled(sched, K_inv_tiles, y_tiles, alpha_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Launch asynchronous loss computation where + // loss(theta) = 0.5 * ( log(det(K)) - y^T * K^-1 * y - N * log(2 * pi) ) + auto loss_value = compute_loss_tiled(sched, K_tiles, alpha_tiles, y_tiles, n_tile_size, n_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Launch asynchronous update of the hyperparameters + if (trainable_params[0]) + { // lengthscale + update_hyperparameter_tiled_lengthscale( + sched, + K_inv_tiles, + grad_l_tiles, + alpha_tiles, + adam_params, + diag_tiles, + inter_alpha, + sek_params, + n_tile_size, + n_tiles, + iter, + 0); + } + if (trainable_params[1]) + { // vertical_lengthscale + update_hyperparameter_tiled_lengthscale( + sched, + K_inv_tiles, + grad_v_tiles, + alpha_tiles, + adam_params, + diag_tiles, + inter_alpha, + sek_params, + n_tile_size, + n_tiles, + iter, + 1); + } + if (trainable_params[2]) + { // noise_variance + update_hyperparameter_tiled_noise_variance( + sched, K_inv_tiles, alpha_tiles, adam_params, sek_params, n_tile_size, n_tiles, iter, 2); + } + // Synchronize after iteration + losses.push_back(loss_value.get()); + } + return losses; +} /** * @brief Perform a single optimization step @@ -167,15 +1126,17 @@ optimize(const std::vector &training_input, * @param n_tile_size The size of each training tile * @param n_regressors The number of regressors * - * @param hyperparams The Adam optimizer hyperparameters - * @param hyperparameters The kernel hyperparameters - * @param trainable_params The vector containing a bool wheather to train a hyperparameter + * @param adam_params The Adam optimizer hyperparameters + * @param sek_params The kernel hyperparameters + * @param trainable_params The vector containing a bool whether to train a hyperparameter * * @param iter The current optimization iteration * * @return The loss value */ -double optimize_step(const std::vector &training_input, +template +double optimize_step(Scheduler &sched, + const std::vector &training_input, const std::vector &training_output, std::size_t n_tiles, std::size_t n_tile_size, @@ -183,7 +1144,25 @@ double optimize_step(const std::vector &training_input, AdamParams &adam_params, SEKParams &sek_params, std::vector trainable_params, - std::size_t iter); + std::size_t iter) +{ + // No point in copy&pasting everything for this function + const auto old_opt_iter = adam_params.opt_iter; + adam_params.opt_iter = iter + 1; + const auto r = optimize( + sched, + training_input, + training_output, + n_tiles, + n_tile_size, + n_regressors, + adam_params, + sek_params, + trainable_params, + iter); + adam_params.opt_iter = old_opt_iter; + return r[0]; +} } // end of namespace cpu diff --git a/core/include/gprat/cpu/tiled_algorithms.hpp b/core/include/gprat/cpu/tiled_algorithms.hpp index a8706fe9..5cec2db5 100644 --- a/core/include/gprat/cpu/tiled_algorithms.hpp +++ b/core/include/gprat/cpu/tiled_algorithms.hpp @@ -3,32 +3,99 @@ #pragma once +#include "gprat/cpu/adapter_cblas_fp64.hpp" +#include "gprat/cpu/gp_algorithms.hpp" +#include "gprat/cpu/gp_optimizer.hpp" +#include "gprat/cpu/gp_uncertainty.hpp" +#include "gprat/detail/async_helpers.hpp" #include "gprat/detail/config.hpp" #include "gprat/hyperparameters.hpp" #include "gprat/kernels.hpp" -#include "gprat/tile_data.hpp" +#include "gprat/scheduler.hpp" #include GPRAT_NS_BEGIN -using Tiled_matrix = std::vector>>; -using Tiled_vector = std::vector>>; - namespace cpu { +namespace impl +{ +void update_parameters( + const AdamParams &adam_params, + SEKParams &sek_params, + std::size_t N, + std::size_t n_tiles, + std::size_t iter, + std::size_t param_idx, + double trace, + double dot, + bool jitter, + double factor); +} + // Tiled Cholesky Algorithm /** * @brief Perform right-looking tiled Cholesky decomposition. * - * @param ft_tiles Tiled matrix represented as a vector of futurized tiles, containing the + * @param tiles Tiled matrix represented as a vector of futurized tiles, containing the * covariance matrix, afterwards the Cholesky decomposition. * @param N Tile size per dimension. * @param n_tiles Number of tiles per dimension. */ -void right_looking_cholesky_tiled(Tiled_matrix &ft_tiles, std::size_t N, std::size_t n_tiles); +template +void right_looking_cholesky_tiled(Scheduler &sched, Tiles &tiles, std::size_t N, std::size_t n_tiles) +{ + for (std::size_t k = 0; k < n_tiles; k++) + { + // POTRF: Compute Cholesky factor L + tiles[k * n_tiles + k] = detail::named_dataflow( + sched, schedule::cholesky_potrf(sched, n_tiles, k), "cholesky_tiled", tiles[k * n_tiles + k], N); + for (std::size_t m = k + 1; m < n_tiles; m++) + { + // TRSM: Solve X * L^T = A + tiles[m * n_tiles + k] = detail::named_dataflow( + sched, + schedule::cholesky_trsm(sched, n_tiles, k, m), + "cholesky_tiled", + tiles[k * n_tiles + k], + tiles[m * n_tiles + k], + N, + N, + Blas_trans, + Blas_right); + } + for (std::size_t m = k + 1; m < n_tiles; m++) + { + // SYRK: A = A - B * B^T + tiles[m * n_tiles + m] = detail::named_dataflow( + sched, + schedule::cholesky_syrk(sched, n_tiles, m), + "cholesky_tiled", + tiles[m * n_tiles + m], + tiles[m * n_tiles + k], + N); + for (std::size_t n = k + 1; n < m; n++) + { + // GEMM: C = C - A * B^T + tiles[m * n_tiles + n] = detail::named_dataflow( + sched, + schedule::cholesky_gemm(sched, n_tiles, k, m, n), + "cholesky_tiled", + tiles[m * n_tiles + k], + tiles[n * n_tiles + k], + tiles[m * n_tiles + n], + N, + N, + N, + Blas_no_trans, + Blas_trans); + } + } + } +} // Tiled Triangular Solve Algorithms @@ -40,7 +107,37 @@ void right_looking_cholesky_tiled(Tiled_matrix &ft_tiles, std::size_t N, std::si * @param N Tile size per dimension. * @param n_tiles Number of tiles per dimension. */ -void forward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, std::size_t N, std::size_t n_tiles); +template +void forward_solve_tiled(Scheduler &sched, Tiles &ft_tiles, Tiles &ft_rhs, std::size_t N, std::size_t n_tiles) +{ + for (std::size_t k = 0; k < n_tiles; k++) + { + // TRSM: Solve L * x = a + ft_rhs[k] = detail::named_dataflow( + sched, + schedule::solve_trsv(sched, n_tiles, k), + "triangular_solve_tiled", + ft_tiles[k * n_tiles + k], + ft_rhs[k], + N, + Blas_no_trans); + for (std::size_t m = k + 1; m < n_tiles; m++) + { + // GEMV: b = b - A * a + ft_rhs[m] = detail::named_dataflow( + sched, + schedule::solve_gemv(sched, n_tiles, k, m), + "triangular_solve_tiled", + ft_tiles[m * n_tiles + k], + ft_rhs[k], + ft_rhs[m], + N, + N, + Blas_substract, + Blas_no_trans); + } + } +} /** * @brief Perform tiled backward triangular matrix-vector solve. @@ -50,7 +147,39 @@ void forward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, std::size * @param N Tile size per dimension. * @param n_tiles Number of tiles per dimension. */ -void backward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, std::size_t N, std::size_t n_tiles); +template +void backward_solve_tiled(Scheduler &sched, Tiles &ft_tiles, Tiles &ft_rhs, std::size_t N, std::size_t n_tiles) +{ + for (int k_ = static_cast(n_tiles) - 1; k_ >= 0; k_--) // int instead of std::size_t for last comparison + { + std::size_t k = static_cast(k_); + // TRSM: Solve L^T * x = a + ft_rhs[k] = detail::named_dataflow( + sched, + schedule::solve_trsm(sched, n_tiles, k), + "triangular_solve_tiled", + ft_tiles[k * n_tiles + k], + ft_rhs[k], + N, + Blas_trans); + for (int m_ = k_ - 1; m_ >= 0; m_--) // int instead of std::size_t for last comparison + { + std::size_t m = static_cast(m_); + // GEMV:b = b - A^T * a + ft_rhs[m] = detail::named_dataflow( + sched, + schedule::solve_gemv(sched, n_tiles, k, m), + "triangular_solve_tiled", + ft_tiles[k * n_tiles + m], + ft_rhs[k], + ft_rhs[m], + N, + N, + Blas_substract, + Blas_trans); + } + } +} /** * @brief Perform tiled forward triangular matrix-matrix solve. @@ -62,12 +191,50 @@ void backward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, std::siz * @param n_tiles Number of tiles in first dimension. * @param m_tiles Number of tiles in second dimension. */ -void forward_solve_tiled_matrix(Tiled_matrix &ft_tiles, - Tiled_matrix &ft_rhs, - std::size_t N, - std::size_t M, - std::size_t n_tiles, - std::size_t m_tiles); +template +void forward_solve_tiled_matrix( + Scheduler &sched, + Tiles &ft_tiles, + Tiles &ft_rhs, + std::size_t N, + std::size_t M, + std::size_t n_tiles, + std::size_t m_tiles) +{ + for (std::size_t c = 0; c < m_tiles; c++) + { + for (std::size_t k = 0; k < n_tiles; k++) + { + // TRSM: solve L * X = A + ft_rhs[k * m_tiles + c] = detail::named_dataflow( + sched, + schedule::solve_matrix_trsm(sched, m_tiles, c, k), + "triangular_solve_tiled_matrix", + ft_tiles[k * n_tiles + k], + ft_rhs[k * m_tiles + c], + N, + M, + Blas_no_trans, + Blas_left); + for (std::size_t m = k + 1; m < n_tiles; m++) + { + // GEMM: C = C - A * B + ft_rhs[m * m_tiles + c] = detail::named_dataflow( + sched, + schedule::solve_matrix_gemm(sched, m_tiles, c, k, m), + "triangular_solve_tiled_matrix", + ft_tiles[m * n_tiles + k], + ft_rhs[k * m_tiles + c], + ft_rhs[m * m_tiles + c], + N, + M, + N, + Blas_no_trans, + Blas_no_trans); + } + } + } +} /** * @brief Perform tiled backward triangular matrix-matrix solve. @@ -79,12 +246,52 @@ void forward_solve_tiled_matrix(Tiled_matrix &ft_tiles, * @param n_tiles Number of tiles in first dimension. * @param m_tiles Number of tiles in second dimension. */ -void backward_solve_tiled_matrix(Tiled_matrix &ft_tiles, - Tiled_matrix &ft_rhs, - std::size_t N, - std::size_t M, - std::size_t n_tiles, - std::size_t m_tiles); +template +void backward_solve_tiled_matrix( + Scheduler &sched, + Tiles &ft_tiles, + Tiles &ft_rhs, + std::size_t N, + std::size_t M, + std::size_t n_tiles, + std::size_t m_tiles) +{ + for (std::size_t c = 0; c < m_tiles; c++) + { + for (int k_ = static_cast(n_tiles) - 1; k_ >= 0; k_--) // int instead of std::size_t for last comparison + { + std::size_t k = static_cast(k_); + // TRSM: solve L^T * X = A + ft_rhs[k * m_tiles + c] = detail::named_dataflow( + sched, + schedule::solve_matrix_trsm(sched, m_tiles, c, k), + "triangular_solve_tiled_matrix", + ft_tiles[k * n_tiles + k], + ft_rhs[k * m_tiles + c], + N, + M, + Blas_trans, + Blas_left); + for (int m_ = k_ - 1; m_ >= 0; m_--) // int instead of std::size_t for last comparison + { + std::size_t m = static_cast(m_); + // GEMM: C = C - A^T * B + ft_rhs[m * m_tiles + c] = detail::named_dataflow( + sched, + schedule::solve_matrix_gemm(sched, m_tiles, c, k, m), + "triangular_solve_tiled_matrix", + ft_tiles[k * n_tiles + m], + ft_rhs[k * m_tiles + c], + ft_rhs[m * m_tiles + c], + N, + M, + N, + Blas_trans, + Blas_no_trans); + } + } + } +} /** * @brief Perform tiled matrix-vector multiplication @@ -97,13 +304,34 @@ void backward_solve_tiled_matrix(Tiled_matrix &ft_tiles, * @param n_tiles Number of tiles in first dimension. * @param m_tiles Number of tiles in second dimension. */ -void matrix_vector_tiled(Tiled_matrix &ft_tiles, - Tiled_vector &ft_vector, - Tiled_vector &ft_rhs, +template +void matrix_vector_tiled(Scheduler &sched, + Tiles &ft_tiles, + Tiles &ft_vector, + Tiles &ft_rhs, std::size_t N_row, std::size_t N_col, std::size_t n_tiles, - std::size_t m_tiles); + std::size_t m_tiles) +{ + for (std::size_t k = 0; k < m_tiles; k++) + { + for (std::size_t m = 0; m < n_tiles; m++) + { + ft_rhs[k] = detail::named_dataflow( + sched, + schedule::multiply_gemv(sched, n_tiles, k, m), + "prediction_tiled", + ft_tiles[k * n_tiles + m], + ft_vector[m], + ft_rhs[k], + N_row, + N_col, + Blas_add, + Blas_no_trans); + } + } +} /** * @brief Perform tiled symmetric k-rank update on diagonal tiles @@ -115,13 +343,33 @@ void matrix_vector_tiled(Tiled_matrix &ft_tiles, * @param n_tiles Number of tiles in first dimension. * @param m_tiles Number of tiles in second dimension. */ +template void symmetric_matrix_matrix_diagonal_tiled( - Tiled_matrix &ft_tiles, - Tiled_vector &ft_vector, + Scheduler &sched, + Tiles &ft_tiles, + Tiles &ft_vector, std::size_t N, std::size_t M, std::size_t n_tiles, - std::size_t m_tiles); + std::size_t m_tiles) +{ + for (std::size_t i = 0; i < m_tiles; ++i) + { + for (std::size_t n = 0; n < n_tiles; ++n) + { + // Compute inner product to obtain diagonal elements of + // V^T * V <=> cross(K) * K^-1 * cross(K)^T + ft_vector[i] = detail::named_dataflow( + sched, + schedule::k_rank_dot_diag_syrk(sched, m_tiles, i), + "posterior_tiled", + ft_tiles[n * m_tiles + i], + ft_vector[i], + N, + M); + } + } +} /** * @brief Perform tiled symmetric k-rank update (ft_tiles^T * ft_tiles) @@ -133,12 +381,40 @@ void symmetric_matrix_matrix_diagonal_tiled( * @param n_tiles Number of tiles in first dimension. * @param m_tiles Number of tiles in second dimension. */ -void symmetric_matrix_matrix_tiled(Tiled_matrix &ft_tiles, - Tiled_matrix &ft_result, - std::size_t N, - std::size_t M, - std::size_t n_tiles, - std::size_t m_tiles); +template +void symmetric_matrix_matrix_tiled( + Scheduler &sched, + Tiles &ft_tiles, + Tiles &ft_result, + std::size_t N, + std::size_t M, + std::size_t n_tiles, + std::size_t m_tiles) +{ + for (std::size_t c = 0; c < m_tiles; c++) + { + for (std::size_t k = 0; k < m_tiles; k++) + { + for (std::size_t m = 0; m < n_tiles; m++) + { + // (SYRK for (c == k) possible) + // GEMM: C = C - A^T * B + ft_result[c * m_tiles + k] = detail::named_dataflow( + sched, + schedule::k_rank_gemm(sched, m_tiles, c, k, m), + "triangular_solve_tiled_matrix", + ft_tiles[m * m_tiles + c], + ft_tiles[m * m_tiles + k], + ft_result[c * m_tiles + k], + N, + M, + M, + Blas_trans, + Blas_no_trans); + } + } + } +} /** * @brief Compute the difference between two tiled vectors @@ -147,7 +423,16 @@ void symmetric_matrix_matrix_tiled(Tiled_matrix &ft_tiles, * @param M Tile size dimension. * @param m_tiles Number of tiles. */ -void vector_difference_tiled(Tiled_vector &ft_minuend, Tiled_vector &ft_subtrahend, std::size_t M, std::size_t m_tiles); +template +void vector_difference_tiled( + Scheduler &sched, Tiles &ft_minuend, Tiles &ft_subtrahend, std::size_t M, std::size_t m_tiles) +{ + for (std::size_t i = 0; i < m_tiles; i++) + { + ft_subtrahend[i] = detail::named_dataflow( + sched, schedule::vector_axpy(sched, m_tiles, i), "uncertainty_tiled", ft_minuend[i], ft_subtrahend[i], M); + } +} /** * @brief Extract the tiled diagonals of a tiled matrix @@ -156,7 +441,15 @@ void vector_difference_tiled(Tiled_vector &ft_minuend, Tiled_vector &ft_subtrahe * @param M Tile size per dimension. * @param m_tiles Number of tiles per dimension. */ -void matrix_diagonal_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_vector, std::size_t M, std::size_t m_tiles); +template +void matrix_diagonal_tiled(Scheduler &sched, Tiles &ft_tiles, Tiles &ft_vector, std::size_t M, std::size_t m_tiles) +{ + for (std::size_t i = 0; i < m_tiles; i++) + { + ft_vector[i] = detail::named_dataflow( + sched, schedule::get_diagonal(sched, m_tiles, i), "uncertainty_tiled", ft_tiles[i * m_tiles + i], M); + } +} /** * @brief Compute the negative log likelihood loss with a tiled covariance matrix K. @@ -165,17 +458,30 @@ void matrix_diagonal_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_vector, std: * * @param ft_tiles Tiled Cholesky factor matrix represented as a vector of futurized tiles. * @param ft_alpha Tiled vector containing the solution of K^-1 * y - * @param ft_y Tiled vector containing the the training output y - * @param loss The loss value to be computed + * @param ft_y Tiled vector containing the training output y * @param N Tile size per dimension. * @param n_tiles Number of tiles per dimension. + * @return The loss value to be computed */ -void compute_loss_tiled(Tiled_matrix &ft_tiles, - Tiled_vector &ft_alpha, - Tiled_vector &ft_y, - hpx::shared_future &loss, - std::size_t N, - std::size_t n_tiles); +template +hpx::future +compute_loss_tiled(Scheduler &sched, Tiles &ft_tiles, Tiles &ft_alpha, Tiles &ft_y, std::size_t N, std::size_t n_tiles) +{ + std::vector> loss_tiled; + loss_tiled.reserve(n_tiles); + for (std::size_t k = 0; k < n_tiles; k++) + { + loss_tiled.push_back(detail::named_dataflow( + sched, + schedule::compute_loss(sched, n_tiles, k), + "loss_tiled", + ft_tiles[k * n_tiles + k], + ft_alpha[k], + ft_y[k], + N)); + } + return detail::named_dataflow("loss_tiled", loss_tiled, N, n_tiles); +} /** * @brief Updates a hyperparameter of the SEK kernel using Adam @@ -190,16 +496,157 @@ void compute_loss_tiled(Tiled_matrix &ft_tiles, * @param iter Current iteration. * @param param_idx Index of the hyperparameter to optimize. */ -void update_hyperparameter_tiled( - const Tiled_matrix &ft_invK, - const Tiled_matrix &ft_gradK_param, - const Tiled_vector &ft_alpha, +template +void update_hyperparameter_tiled_lengthscale( + Scheduler &sched, + const Tiles &ft_invK, + const Tiles &ft_gradK_param, + const Tiles &ft_alpha, const AdamParams &adam_params, + Tiles &diag_tiles, // Diagonal tiles + Tiles &inter_alpha, // Intermediate result SEKParams &sek_params, std::size_t N, std::size_t n_tiles, std::size_t iter, - std::size_t param_idx); + std::size_t param_idx) +{ + /* + * PART 1: + * Compute gradient = 0.5 * ( trace(inv(K) * grad(K)_param) + y^T * inv(K) * grad(K)_param * inv(K) * y ) + * + * 1: Compute trace(inv(K) * grad(K)_param) + * 2: Compute y^T * inv(K) * grad(K)_param * inv(K) * y + * + * Update parameter: + * 3: Update moments + * - m_T = beta1 * m_T-1 + (1 - beta1) * g_T + * - w_T = beta2 + w_T-1 + (1 - beta2) * g_T^2 + * 4: Adam step: + * - nu_T = nu * sqrt(1 - beta2_T) / (1 - beta1_T) + * - theta_T = theta_T-1 - nu_T * m_T / (sqrt(w_T) + epsilon) + */ + hpx::shared_future trace = hpx::make_ready_future(0.0); + hpx::shared_future dot = hpx::make_ready_future(0.0); + bool jitter = false; + double factor = 1.0; + + // Reset our helper tiles + for (std::size_t d = 0; d < n_tiles; d++) + { + diag_tiles[d] = detail::named_make_tile( + sched, schedule::diag_tile(sched, n_tiles, d), "assemble", diag_tiles[d], N); + inter_alpha[d] = detail::named_make_tile( + sched, schedule::inter_alpha_tile(sched, n_tiles, d), "assemble", inter_alpha[d], N); + } + + //////////////////////////////////// + // PART 1: Compute gradient + // Step 1: Compute trace(inv(K)*grad_K_param) + // Compute diagonal tiles of inv(K) * grad(K)_param + for (std::size_t i = 0; i < n_tiles; ++i) + { + for (std::size_t j = 0; j < n_tiles; ++j) + { + diag_tiles[i] = detail::named_dataflow( + sched, + schedule::diag_tile(sched, n_tiles, i), + "trace", + ft_invK[i * n_tiles + j], + ft_gradK_param[j * n_tiles + i], + diag_tiles[i], + N, + N); + } + } + // Compute the trace of the diagonal tiles + for (std::size_t j = 0; j < n_tiles; ++j) + { + trace = detail::named_dataflow( + sched, schedule::diag_tile(sched, n_tiles, j), "trace", diag_tiles[j], trace); + } + // Not sure if can be done this way + // Step 2: Compute alpha^T * grad(K)_param * alpha (with alpha = inv(K) * y) + // Compute inter_alpha = grad(K)_param * alpha + for (std::size_t k = 0; k < n_tiles; k++) + { + for (std::size_t m = 0; m < n_tiles; m++) + { + inter_alpha[k] = detail::named_dataflow( + sched, + schedule::inter_alpha_tile(sched, n_tiles, k), + "gemv", + ft_gradK_param[k * n_tiles + m], + ft_alpha[m], + inter_alpha[k], + N, + N, + Blas_add, + Blas_no_trans); + } + } + // Compute alpha^T * inter_alpha + for (std::size_t j = 0; j < n_tiles; ++j) + { + dot = detail::named_dataflow( + sched, schedule::inter_alpha_tile(sched, n_tiles, j), "grad_right_tiled", inter_alpha[j], ft_alpha[j], dot); + } + + impl::update_parameters( + adam_params, sek_params, N, n_tiles, iter, param_idx, trace.get(), dot.get(), jitter, factor); +} + +template +void update_hyperparameter_tiled_noise_variance( + Scheduler &sched, + const Tiles &ft_invK, + const Tiles &ft_alpha, + const AdamParams &adam_params, + SEKParams &sek_params, + std::size_t N, + std::size_t n_tiles, + std::size_t iter, + std::size_t param_idx) +{ + /* + * PART 1: + * Compute gradient = 0.5 * ( trace(inv(K) * grad(K)_param) + y^T * inv(K) * grad(K)_param * inv(K) * y ) + * + * 1: Compute trace(inv(K) * grad(K)_param) + * 2: Compute y^T * inv(K) * grad(K)_param * inv(K) * y + * + * Update parameter: + * 3: Update moments + * - m_T = beta1 * m_T-1 + (1 - beta1) * g_T + * - w_T = beta2 + w_T-1 + (1 - beta2) * g_T^2 + * 4: Adam step: + * - nu_T = nu * sqrt(1 - beta2_T) / (1 - beta1_T) + * - theta_T = theta_T-1 - nu_T * m_T / (sqrt(w_T) + epsilon) + */ + hpx::shared_future trace = hpx::make_ready_future(0.0); + hpx::shared_future dot = hpx::make_ready_future(0.0); + bool jitter = true; + double factor = 1.0; + + //////////////////////////////////// + // PART 1: Compute gradient + // Step 1: Compute the trace of inv(K) * noise_variance + for (std::size_t j = 0; j < n_tiles; ++j) + { + trace = detail::named_dataflow(sched, schedule::K_inv_tile(sched, n_tiles, j, j), "grad_left_tiled", ft_invK[j * n_tiles + j], trace, N); + } + //////////////////////////////////// + // Step 2: Compute the alpha^T * alpha * noise_variance + for (std::size_t j = 0; j < n_tiles; ++j) + { + dot = detail::named_dataflow(sched, schedule::alpha_tile(sched, n_tiles, j),"grad_right_tiled", ft_alpha[j], ft_alpha[j], dot); + } + + factor = compute_sigmoid(to_unconstrained(sek_params.noise_variance, true)); + + impl::update_parameters( + adam_params, sek_params, N, n_tiles, iter, param_idx, trace.get(), dot.get(), jitter, factor); +} } // end of namespace cpu diff --git a/core/include/gprat/detail/async_helpers.hpp b/core/include/gprat/detail/async_helpers.hpp index b04ef144..05a24a91 100644 --- a/core/include/gprat/detail/async_helpers.hpp +++ b/core/include/gprat/detail/async_helpers.hpp @@ -5,15 +5,26 @@ #include "gprat/detail/config.hpp" +#include #include #include #include GPRAT_NS_BEGIN +/// @brief Empty type representing local scheduling (always on this locality) +struct basic_local_scheduler +{ }; + namespace detail { +// Functions prefixed with named_* allow the user to specify a custom name for this entry in the +// execution graph. Much like wrapping your function with hpx::annotated_function would. + +// ============================================================= +// non-scheduler aware + template decltype(auto) named_dataflow(const char *name, Args &&...args) { @@ -26,6 +37,36 @@ decltype(auto) named_async(const char *name, Args &&...args) return hpx::async(hpx::annotated_function(F, name), std::forward(args)...); } +// ============================================================= +// local shared-memory scheduling +// (no-op, same as above) + +template +decltype(auto) named_make_tile(const basic_local_scheduler & /*sched*/, + std::size_t /*on*/, + const char *name, + TileReference & /*target*/, + Args &&...args) +{ + // This method basically ignores the reference to the target tile as the non-action factories don't need it. + // (They always create the tile_data locally and return that - only the HPX action wrappers need a reference) + return hpx::dataflow(hpx::annotated_function(hpx::unwrapping(F), name), std::forward(args)...); +} + +template +decltype(auto) +named_dataflow(const basic_local_scheduler & /*sched*/, std::size_t /*on*/, const char *name, Args &&...args) +{ + return hpx::dataflow(hpx::annotated_function(hpx::unwrapping(F), name), std::forward(args)...); +} + +template +decltype(auto) +named_async(const basic_local_scheduler & /*sched*/, std::size_t /*on*/, const char *name, Args &&...args) +{ + return hpx::async(hpx::annotated_function(F, name), std::forward(args)...); +} + } // namespace detail GPRAT_NS_END diff --git a/core/include/gprat/scheduler.hpp b/core/include/gprat/scheduler.hpp new file mode 100644 index 00000000..e19af509 --- /dev/null +++ b/core/include/gprat/scheduler.hpp @@ -0,0 +1,89 @@ +#ifndef GPRAT_CPU_SCHEDULER_HPP +#define GPRAT_CPU_SCHEDULER_HPP + +#pragma once + +#include "gprat/detail/async_helpers.hpp" + +// TODO: move to separate header +#include "gprat/tile_data.hpp" +#include +#include + +GPRAT_NS_BEGIN + +using tiled_scheduler_local = basic_local_scheduler; + +template +using tiled_dataset_local = std::vector>>; + +template +struct tile_dataset_type; + +template +struct tile_dataset_type +{ + using type = tiled_dataset_local; +}; + +template +tiled_dataset_local +make_tiled_dataset(const tiled_scheduler_local &, std::size_t num_tiles, Mapper &&) +{ + return std::vector>>{ num_tiles }; +} + +/// @brief This namespace contains the operation placement functions for all schedulers. +namespace schedule { + +#ifdef _MSC_VER +#pragma warning(push) +#pragma warning(disable:4100) +#endif + +// ============================================================= +// local scheduler + +constexpr std::size_t covariance_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return 0; } +constexpr std::size_t cross_covariance_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return 0; } +constexpr std::size_t alpha_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t i) { return 0; } +constexpr std::size_t prediction_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t i) { return 0; } +constexpr std::size_t t_cross_covariance_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return 0; } +constexpr std::size_t prior_K_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return 0; } +constexpr std::size_t K_inv_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return 0; } +constexpr std::size_t K_grad_v_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return 0; } +constexpr std::size_t K_grad_l_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return 0; } +constexpr std::size_t uncertainty_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t i) { return 0; } +constexpr std::size_t inter_alpha_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t i) { return 0; } +constexpr std::size_t diag_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t i) { return 0; } + +constexpr std::size_t cholesky_potrf(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k) { return 0; } +constexpr std::size_t cholesky_syrk(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t m) { return 0; } +constexpr std::size_t cholesky_trsm(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k, std::size_t m) { return 0; } +constexpr std::size_t cholesky_gemm(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k, std::size_t m, std::size_t n) { return 0; } + +constexpr std::size_t solve_trsv(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k) { return 0; } +constexpr std::size_t solve_trsm(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k) { return 0; } +constexpr std::size_t solve_gemv(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k, std::size_t m) { return 0; } + +constexpr std::size_t solve_matrix_trsm(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t c,std::size_t k) { return 0; } +constexpr std::size_t solve_matrix_gemm(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t c,std::size_t k, std::size_t m) { return 0; } + +constexpr std::size_t multiply_gemv(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k, std::size_t m) { return 0; } + +constexpr std::size_t k_rank_dot_diag_syrk(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k) { return 0; } +constexpr std::size_t k_rank_gemm(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t c,std::size_t k, std::size_t m) { return 0; } + +constexpr std::size_t vector_axpy(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k) { return 0; } +constexpr std::size_t get_diagonal(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k) { return 0; } +constexpr std::size_t compute_loss(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k) { return 0; } + +#ifdef _MSC_VER +#pragma warning(pop) +#endif + +} + +GPRAT_NS_END + +#endif diff --git a/core/src/cpu/gp_functions.cpp b/core/src/cpu/gp_functions.cpp index 0e32eac7..097f4867 100644 --- a/core/src/cpu/gp_functions.cpp +++ b/core/src/cpu/gp_functions.cpp @@ -14,913 +14,6 @@ namespace cpu /////////////////////////////////////////////////////////////////////////// // PREDICT -std::vector> -cholesky(const std::vector &training_input, - const SEKParams &sek_params, - std::size_t n_tiles, - std::size_t n_tile_size, - std::size_t n_regressors) -{ - std::vector> result; - // Tiled future data structures - Tiled_matrix K_tiles; // Tiled covariance matrix - - // Preallocate memory - result.resize(n_tiles * n_tiles); - K_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous assembly - for (std::size_t i = 0; i < n_tiles; i++) - { - for (std::size_t j = 0; j <= i; j++) - { - K_tiles[i * n_tiles + j] = detail::named_async( - "assemble_tiled_K", i, j, n_tile_size, n_regressors, sek_params, training_input); - } - } - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(K_tiles, n_tile_size, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Synchronize - for (std::size_t i = 0; i < n_tiles; i++) - { - for (std::size_t j = 0; j <= i; j++) - { - result[i * n_tiles + j] = K_tiles[i * n_tiles + j].get(); - } - } - return result; -} - -std::vector -predict(const std::vector &training_input, - const std::vector &training_output, - const std::vector &test_input, - const SEKParams &sek_params, - std::size_t n_tiles, - std::size_t n_tile_size, - std::size_t m_tiles, - std::size_t m_tile_size, - std::size_t n_regressors) -{ - /* - * Prediction: hat(y)_M = cross(K)_MxN * K^-1_NxN * y_N - * - Covariance matrix K_NxN - * - Cross-covariance cross(K)_MxN - * - Training ouput y_N - * - Prediction output hat(y)_M - * - * Algorithm: - * 1: Compute lower triangular part of covariance matrix K - * 2: Compute Cholesky factor L of K - * 3: Compute prediction hat(y): - * - triangular solve L * beta = y - * - triangular solve L^T * alpha = beta - * - compute hat(y) = cross(K) * alpha - */ - - std::vector prediction_result; - // Tiled future data structures - Tiled_matrix K_tiles; // Tiled covariance matrix - Tiled_matrix cross_covariance_tiles; // Tiled cross_covariance matrix - Tiled_vector prediction_tiles; // Tiled solution - Tiled_vector alpha_tiles; // Tiled intermediate solution - - // Preallocate memory - prediction_result.reserve(test_input.size()); - - K_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure - alpha_tiles.reserve(n_tiles); - cross_covariance_tiles.reserve(m_tiles * n_tiles); - prediction_tiles.reserve(m_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous assembly - for (std::size_t i = 0; i < n_tiles; i++) - { - for (std::size_t j = 0; j <= i; j++) - { - K_tiles[i * n_tiles + j] = detail::named_async( - "assemble_tiled_K", i, j, n_tile_size, n_regressors, sek_params, training_input); - } - } - - for (std::size_t i = 0; i < n_tiles; i++) - { - alpha_tiles.push_back( - detail::named_async("assemble_tiled_alpha", i, n_tile_size, training_output)); - } - - for (std::size_t i = 0; i < m_tiles; i++) - { - for (std::size_t j = 0; j < n_tiles; j++) - { - cross_covariance_tiles.push_back(detail::named_async( - "assemble_pred", i, j, m_tile_size, n_tile_size, n_regressors, sek_params, test_input, training_input)); - } - } - - for (std::size_t i = 0; i < m_tiles; i++) - { - prediction_tiles.push_back(detail::named_async("assemble_tiled", m_tile_size)); - } - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(K_tiles, n_tile_size, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous triangular solve L * (L^T * alpha) = y - forward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); - backward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous prediction computation solve: \hat{y} = K_cross_cov * alpha - matrix_vector_tiled( - cross_covariance_tiles, alpha_tiles, prediction_tiles, m_tile_size, n_tile_size, n_tiles, m_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Synchronize prediction - for (std::size_t i = 0; i < m_tiles; i++) - { - auto tile = prediction_tiles[i].get(); - std::copy_n(tile.data(), tile.size(), std::back_inserter(prediction_result)); - } - return prediction_result; -} - -std::vector> predict_with_uncertainty( - const std::vector &training_input, - const std::vector &training_output, - const std::vector &test_input, - const SEKParams &sek_params, - std::size_t n_tiles, - std::size_t n_tile_size, - std::size_t m_tiles, - std::size_t m_tile_size, - std::size_t n_regressors) -{ - /* - * Prediction: hat(y) = cross(K) * K^-1 * y - * Uncertainty: diag(Sigma) = diag(prior(K)) * diag(cross(K)^T * K^-1 * cross(K)) - * - Covariance matrix K_NxN - * - Cross-covariance cross(K)_MxN - * - Prior covariance prior(K)_MxM - * - Training ouput y_N - * - Prediction output hat(y)_M - * - Posterior covariance matrix Sigma_MxM - * - * Algorithm: - * 1: Compute lower triangular part of covariance matrix K - * 2: Compute Cholesky factor L of K - * 3: Compute prediction hat(y): - * - triangular solve L * beta = y - * - triangular solve L^T * alpha = beta - * - compute hat(y) = cross(K) * alpha - * 4: Compute uncertainty diag(Sigma): - * - triangular solve L * V = cross(K)^T - * - compute diag(W) = diag(V^T * V) - * - compute diag(Sigma) = diag(prior(K)) - diag(W) - */ - - std::vector prediction_result; - std::vector uncertainty_result; - // Tiled future data structures for prediction - Tiled_matrix K_tiles; // Tiled covariance matrix K_NxN - Tiled_matrix cross_covariance_tiles; // Tiled cross_covariance matrix K_NxM - Tiled_vector prediction_tiles; // Tiled solution - Tiled_vector alpha_tiles; // Tiled intermediate solution - // Tiled future data structures for uncertainty - Tiled_matrix t_cross_covariance_tiles; // Tiled transposed cross_covariance matrix K_MxN - Tiled_vector prior_K_tiles; // Tiled prior covariance matrix diagonal diag(K_MxM) - Tiled_vector uncertainty_tiles; // Tiled uncertainty solution - - // Preallocate memory - prediction_result.reserve(test_input.size()); - uncertainty_result.reserve(test_input.size()); - - K_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure - cross_covariance_tiles.reserve(m_tiles * n_tiles); - prediction_tiles.reserve(m_tiles); - alpha_tiles.reserve(n_tiles); - - t_cross_covariance_tiles.reserve(n_tiles * m_tiles); - prior_K_tiles.reserve(m_tiles); - uncertainty_tiles.reserve(m_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous assembly - for (std::size_t i = 0; i < n_tiles; i++) - { - for (std::size_t j = 0; j <= i; j++) - { - K_tiles[i * n_tiles + j] = detail::named_async( - "assemble_tiled_K", i, j, n_tile_size, n_regressors, sek_params, training_input); - } - } - - for (std::size_t i = 0; i < n_tiles; i++) - { - alpha_tiles.push_back( - detail::named_async("assemble_tiled_alpha", i, n_tile_size, training_output)); - } - - for (std::size_t i = 0; i < m_tiles; i++) - { - for (std::size_t j = 0; j < n_tiles; j++) - { - cross_covariance_tiles.push_back(detail::named_async( - "assemble_pred", i, j, m_tile_size, n_tile_size, n_regressors, sek_params, test_input, training_input)); - } - } - - for (std::size_t i = 0; i < m_tiles; i++) - { - prediction_tiles.push_back(detail::named_async("assemble_tiled", m_tile_size)); - } - - for (std::size_t i = 0; i < m_tiles; i++) - { - prior_K_tiles.push_back(detail::named_async( - "assemble_tiled", i, i, m_tile_size, n_regressors, sek_params, test_input)); - } - - for (std::size_t j = 0; j < n_tiles; j++) - { - for (std::size_t i = 0; i < m_tiles; i++) - { - t_cross_covariance_tiles.push_back(detail::named_dataflow( - "assemble_pred", m_tile_size, n_tile_size, cross_covariance_tiles[i * n_tiles + j])); - } - } - - for (std::size_t i = 0; i < m_tiles; i++) - { - uncertainty_tiles.push_back(detail::named_async("assemble_prior_inter", m_tile_size)); - } - - // Prediction - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(K_tiles, n_tile_size, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous triangular solve L * (L^T * alpha) = y - forward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); - backward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous prediction computation solve: hat(y) = cross(K) * alpha - matrix_vector_tiled( - cross_covariance_tiles, alpha_tiles, prediction_tiles, m_tile_size, n_tile_size, n_tiles, m_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous triangular solve L * V = cross(K)^T - forward_solve_tiled_matrix(K_tiles, t_cross_covariance_tiles, n_tile_size, m_tile_size, n_tiles, m_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous computation diag(W) = diag(V^T * V) - symmetric_matrix_matrix_diagonal_tiled( - t_cross_covariance_tiles, uncertainty_tiles, n_tile_size, m_tile_size, n_tiles, m_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous computation diag(Sigma) = diag(prior(K)) - diag(W) - vector_difference_tiled(prior_K_tiles, uncertainty_tiles, m_tile_size, m_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Synchronize prediction - for (std::size_t i = 0; i < m_tiles; i++) - { - auto tile = prediction_tiles[i].get(); - std::copy_n(tile.begin(), tile.size(), std::back_inserter(prediction_result)); - } - - // Synchronize uncertainty - for (std::size_t i = 0; i < m_tiles; i++) - { - auto tile = uncertainty_tiles[i].get(); - std::copy_n(tile.begin(), tile.size(), std::back_inserter(uncertainty_result)); - } - - return std::vector>{ std::move(prediction_result), std::move(uncertainty_result) }; -} - -std::vector> predict_with_full_cov( - const std::vector &training_input, - const std::vector &training_output, - const std::vector &test_input, - const SEKParams &sek_params, - std::size_t n_tiles, - std::size_t n_tile_size, - std::size_t m_tiles, - std::size_t m_tile_size, - std::size_t n_regressors) -{ - /* - * Prediction: hat(y)_M = cross(K) * K^-1 * y - * Full covariance: Sigma = prior(K) - cross(K)^T * K^-1 * cross(K) - * - Covariance matrix K_NxN - * - Cross-covariance cross(K)_MxN - * - Prior covariance prior(K)_MxM - * - Training ouput y_N - * - Prediction output hat(y)_M - * - Posterior covariance matrix Sigma_MxM - * - * Algorithm: - * 1: Compute lower triangular part of covariance matrix K - * 2: Compute Cholesky factor L of K - * 3: Compute prediction hat(y): - * - triangular solve L * beta = y - * - triangular solve L^T * alpha = beta - * - compute hat(y) = cross(K) * alpha - * 4: Compute full covariance matrix Sigma: - * - triangular solve L * V = cross(K)^T - * - compute W = V^T * V - * - compute Sigma = prior(K) - W - * 5: Compute diag(Sigma) - */ - - std::vector prediction_result; - std::vector uncertainty_result; - // Tiled future data structures for prediction - Tiled_matrix K_tiles; // Tiled covariance matrix K_NxN - Tiled_matrix cross_covariance_tiles; // Tiled cross_covariance matrix K_NxM - Tiled_vector prediction_tiles; // Tiled solution - Tiled_vector alpha_tiles; // Tiled intermediate solution - // Tiled future data structures for uncertainty - Tiled_matrix t_cross_covariance_tiles; // Tiled transposed cross_covariance matrix K_MxN - Tiled_matrix prior_K_tiles; // Tiled prior covariance matrix K_MxM - Tiled_vector uncertainty_tiles; // Tiled uncertainty solution - - // Preallocate memory - prediction_result.reserve(test_input.size()); - uncertainty_result.reserve(test_input.size()); - - K_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure - cross_covariance_tiles.reserve(m_tiles * n_tiles); - prediction_tiles.reserve(m_tiles); - alpha_tiles.reserve(n_tiles); - - t_cross_covariance_tiles.reserve(n_tiles * m_tiles); - prior_K_tiles.resize(m_tiles * m_tiles); - uncertainty_tiles.reserve(m_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous assembly - for (std::size_t i = 0; i < n_tiles; i++) - { - for (std::size_t j = 0; j <= i; j++) - { - K_tiles[i * n_tiles + j] = detail::named_async( - "assemble_tiled_K", i, j, n_tile_size, n_regressors, sek_params, training_input); - } - } - - for (std::size_t i = 0; i < n_tiles; i++) - { - alpha_tiles.push_back( - detail::named_async("assemble_tiled_alpha", i, n_tile_size, training_output)); - } - - for (std::size_t i = 0; i < m_tiles; i++) - { - for (std::size_t j = 0; j < n_tiles; j++) - { - cross_covariance_tiles.push_back(detail::named_async( - "assemble_pred", i, j, m_tile_size, n_tile_size, n_regressors, sek_params, test_input, training_input)); - } - } - - for (std::size_t i = 0; i < m_tiles; i++) - { - prediction_tiles.push_back(detail::named_async("assemble_tiled", m_tile_size)); - } - - // Assemble prior covariance matrix vector - for (std::size_t i = 0; i < m_tiles; i++) - { - for (std::size_t j = 0; j <= i; j++) - { - prior_K_tiles[i * m_tiles + j] = detail::named_async( - "assemble_prior_tiled", i, j, m_tile_size, n_regressors, sek_params, test_input); - - if (i != j) - { - prior_K_tiles[j * m_tiles + i] = detail::named_dataflow( - "assemble_prior_tiled", m_tile_size, m_tile_size, prior_K_tiles[i * m_tiles + j]); - } - } - } - - for (std::size_t j = 0; j < n_tiles; j++) - { - for (std::size_t i = 0; i < m_tiles; i++) - { - t_cross_covariance_tiles.push_back(detail::named_dataflow( - "assemble_pred", m_tile_size, n_tile_size, cross_covariance_tiles[i * n_tiles + j])); - } - } - - for (std::size_t i = 0; i < m_tiles; i++) - { - uncertainty_tiles.push_back(detail::named_async("assemble_tiled", m_tile_size)); - } - - // Prediction - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(K_tiles, n_tile_size, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous triangular solve L * (L^T * alpha) = y - forward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); - backward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous prediction computation solve: hat(y) = K_cross_cov * alpha - matrix_vector_tiled( - cross_covariance_tiles, alpha_tiles, prediction_tiles, m_tile_size, n_tile_size, n_tiles, m_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous triangular solve L * V = cross(K)^T - forward_solve_tiled_matrix(K_tiles, t_cross_covariance_tiles, n_tile_size, m_tile_size, n_tiles, m_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous computation of full covariance Sigma = prior(K) - V^T * V - symmetric_matrix_matrix_tiled(t_cross_covariance_tiles, prior_K_tiles, n_tile_size, m_tile_size, n_tiles, m_tiles); - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous computation of uncertainty diag(Sigma) - matrix_diagonal_tiled(prior_K_tiles, uncertainty_tiles, m_tile_size, m_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Synchronize prediction - for (std::size_t i = 0; i < m_tiles; i++) - { - auto tile = prediction_tiles[i].get(); - std::copy(tile.begin(), tile.end(), std::back_inserter(prediction_result)); - } - - // Synchronize uncertainty - for (std::size_t i = 0; i < m_tiles; i++) - { - auto tile = uncertainty_tiles[i].get(); - std::copy(tile.begin(), tile.end(), std::back_inserter(uncertainty_result)); - } - - return std::vector>{ std::move(prediction_result), std::move(uncertainty_result) }; -} - -/////////////////////////////////////////////////////////////////////////// -// OPTIMIZATION -double compute_loss(const std::vector &training_input, - const std::vector &training_output, - const SEKParams &sek_params, - std::size_t n_tiles, - std::size_t n_tile_size, - std::size_t n_regressors) -{ - /* - * Negative log likelihood loss: - * loss(theta) = 0.5 * ( log(det(K)) - y^T * K^-1 * y - N * log(2 * pi) ) - * - Covariance matrix K(theta)_NxN - * - Training ouput y_N - * - Hyperparameters theta ={ v, l, v_n } - * - * Algorithm: - * 1: Compute lower triangular part of covariance matrix K - * 2: Compute Cholesky factor L of K - * 3: Compute prediction alpha = K^-1 * y: - * - triangular solve L * beta = y - * - triangular solve L^T * alpha = beta - * 5: Compute beta = K^-1 * y - * 6: Compute negative log likelihood loss - * - Calculate sum_i^N log(L_ii^2) - * - Calculate y^T * beta - * - Add constant N * log (2 * pi) - */ - - hpx::shared_future loss_value; - // Tiled future data structures - Tiled_matrix K_tiles; // Tiled covariance matrix K_NxN - Tiled_vector y_tiles; // Tiled output - Tiled_vector alpha_tiles; // Tiled intermediate solution - - // Preallocate memory - K_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure - y_tiles.reserve(n_tiles); - alpha_tiles.reserve(n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous assembly - for (std::size_t i = 0; i < n_tiles; i++) - { - for (std::size_t j = 0; j <= i; j++) - { - K_tiles[i * n_tiles + j] = detail::named_async( - "assemble_tiled_K", i, j, n_tile_size, n_regressors, sek_params, training_input); - } - } - - for (std::size_t i = 0; i < n_tiles; i++) - { - y_tiles.push_back(detail::named_async("assemble_tiled_y", i, n_tile_size, training_output)); - } - - for (std::size_t i = 0; i < n_tiles; i++) - { - alpha_tiles.push_back( - detail::named_async("assemble_tiled_alpha", i, n_tile_size, training_output)); - } - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(K_tiles, n_tile_size, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous triangular solve L * (L^T * alpha) = y - forward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); - backward_solve_tiled(K_tiles, alpha_tiles, n_tile_size, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous loss computation - compute_loss_tiled(K_tiles, alpha_tiles, y_tiles, loss_value, n_tile_size, n_tiles); - - return loss_value.get(); -} - -std::vector -optimize(const std::vector &training_input, - const std::vector &training_output, - std::size_t n_tiles, - std::size_t n_tile_size, - std::size_t n_regressors, - const AdamParams &adam_params, - SEKParams &sek_params, - std::vector trainable_params) -{ - /* - * - Hyperparameters theta={v, l, v_n} - * - Covariance matrix K(theta) - * - Training ouput y - * - * Algorithm: - * for opt_iter: - * 1: Compute distance for entries of covariance matrix K - * 2: Compute lower triangular part of K with distance - * 3: Compute lower triangular gradients for delta(K)/delta(v), and delta(K)/delta(l) with distance - * - * 4: Compute Cholesky factor L of K - * 5: Compute K^-1: - * - triangular solve L * {} = I - * - triangular solve L^T * K^-1 = {} - * 6: Compute beta = K^-1 * y - * - * 7: Compute negative log likelihood loss - * - Calculate 0.5 sum_i^N log(L_ii^2) - * - Calculate 0.5 y^T * beta - * - Add constant N / 2 * log (2 * pi) - * - * 8: Compute delta(loss)/delta(param_i) - * - Compute trace(K^-1 * delta(K)/delta(theta_i)) - * - Compute beta^T * delta(K)/delta(theta_i) * beta - * 9: Update hyperparameters theta with Adam optimizer - * - m_T = beta1 * m_T-1 + (1 - beta1) * g_T - * - w_T = beta2 + w_T-1 + (1 - beta2) * g_T^2 - * - nu_T = nu * sqrt(1 - beta2_T) / (1 - beta1_T) - * - theta_T = theta_T-1 - nu_T * m_T / (sqrt(w_T) + epsilon) - * endfor - */ - - // data holder for loss - hpx::shared_future loss_value; - // data holder for computed loss values - std::vector losses; - - // Tiled future data structures - Tiled_matrix K_tiles; // Tiled covariance matrix K_NxN - Tiled_vector y_tiles; // Tiled output - Tiled_vector alpha_tiles; // Tiled intermediate solution - Tiled_matrix K_inv_tiles; // Tiled inversed covariance matrix K^-1_NxN - // Tiled future data structures for gradients - Tiled_matrix grad_v_tiles; // Tiled covariance with gradient v - Tiled_matrix grad_l_tiles; // Tiled covariance with gradient l - - // Preallocate memory - losses.reserve(static_cast(adam_params.opt_iter)); - y_tiles.reserve(n_tiles); - - alpha_tiles.resize(n_tiles); // for now resize since reset in loop - K_inv_tiles.resize(n_tiles * n_tiles); // for now resize since reset in loop - - K_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure - grad_v_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure - grad_l_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous assembly of output y - for (std::size_t i = 0; i < n_tiles; i++) - { - y_tiles.push_back(detail::named_async("assemble_y", i, n_tile_size, training_output)); - } - - ////////////////////////////////////////////////////////////////////////////// - // Perform optimization - for (std::size_t iter = 0; iter < static_cast(adam_params.opt_iter); iter++) - { - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous assembly of tiled covariance matrix, derivative of covariance matrix - // vector w.r.t. to vertical lengthscale and derivative of covariance - // matrix vector w.r.t. to lengthscale - for (std::size_t i = 0; i < n_tiles; i++) - { - for (std::size_t j = 0; j <= i; j++) - { - // Compute the distance (z_i - z_j) of K entries to reuse - hpx::shared_future> cov_dists = detail::named_async( - "assemble_cov_dist", i, j, n_tile_size, n_regressors, sek_params, training_input); - - K_tiles[i * n_tiles + j] = detail::named_dataflow( - "assemble_K", i, j, n_tile_size, sek_params, cov_dists); - if (trainable_params[0]) - { - grad_l_tiles[i * n_tiles + j] = - detail::named_dataflow("assemble_gradl", n_tile_size, sek_params, cov_dists); - if (i != j) - { - grad_l_tiles[j * n_tiles + i] = detail::named_dataflow( - "assemble_gradl_t", n_tile_size, n_tile_size, grad_l_tiles[i * n_tiles + j]); - } - } - - if (trainable_params[1]) - { - grad_v_tiles[i * n_tiles + j] = - detail::named_dataflow("assemble_gradv", n_tile_size, sek_params, cov_dists); - if (i != j) - { - grad_v_tiles[j * n_tiles + i] = detail::named_dataflow( - "assemble_gradv_t", n_tile_size, n_tile_size, grad_v_tiles[i * n_tiles + j]); - } - } - } - } - - // Assembly with reallocation -> optimize to only set existing values - for (std::size_t i = 0; i < n_tiles; i++) - { - alpha_tiles[i] = detail::named_async("assemble_tiled", n_tile_size); - } - - for (std::size_t i = 0; i < n_tiles; i++) - { - for (std::size_t j = 0; j < n_tiles; j++) - { - if (i == j) - { - K_inv_tiles[i * n_tiles + j] = - detail::named_async("assemble_identity_matrix", n_tile_size); - } - else - { - K_inv_tiles[i * n_tiles + j] = - detail::named_async("assemble_identity_matrix", n_tile_size * n_tile_size); - } - } - } - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(K_tiles, n_tile_size, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous compute K^-1 through L* (L^T * X) = I - forward_solve_tiled_matrix(K_tiles, K_inv_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); - backward_solve_tiled_matrix(K_tiles, K_inv_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous compute beta = inv(K) * y - matrix_vector_tiled(K_inv_tiles, y_tiles, alpha_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous loss computation where - // loss(theta) = 0.5 * ( log(det(K)) - y^T * K^-1 * y - N * log(2 * pi) ) - compute_loss_tiled(K_tiles, alpha_tiles, y_tiles, loss_value, n_tile_size, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous update of the hyperparameters - if (trainable_params[0]) - { // lengthscale - update_hyperparameter_tiled( - K_inv_tiles, grad_l_tiles, alpha_tiles, adam_params, sek_params, n_tile_size, n_tiles, iter, 0); - } - if (trainable_params[1]) - { // vertical_lengthscale - update_hyperparameter_tiled( - K_inv_tiles, grad_v_tiles, alpha_tiles, adam_params, sek_params, n_tile_size, n_tiles, iter, 1); - } - if (trainable_params[2]) - { // noise_variance - update_hyperparameter_tiled( - K_inv_tiles, - Tiled_matrix{}, // no tiled gradient matrix required - alpha_tiles, - adam_params, - sek_params, - n_tile_size, - n_tiles, - iter, - 2); - } - // Synchronize after iteration - losses.push_back(loss_value.get()); - } - // Return losses - return losses; -} - -double optimize_step(const std::vector &training_input, - const std::vector &training_output, - std::size_t n_tiles, - std::size_t n_tile_size, - std::size_t n_regressors, - AdamParams &adam_params, - SEKParams &sek_params, - std::vector trainable_params, - std::size_t iter) -{ - /* - * - Hyperparameters theta={v, l, v_n} - * - Covariance matrix K(theta) - * - Training ouput y - * - * Algorithm: - * 1: Compute distance for entries of covariance matrix K - * 2: Compute lower triangular part of K with distance - * 3: Compute lower triangular gradients for delta(K)/delta(v), and delta(K)/delta(l) with distance - * - * 4: Compute Cholesky factor L of K - * 5: Compute K^-1: - * - triangular solve L * {} = I - * - triangular solve L^T * K^-1 = {} - * 6: Compute beta = K^-1 * y - * - * 7: Compute negative log likelihood loss - * - Calculate 0.5 sum_i^N log(L_ii^2) - * - Calculate 0.5 y^T * beta - * - Add constant N / 2 * log (2 * pi) - * - * 8: Compute delta(loss)/delta(param_i) - * - Compute trace(K^-1 * delta(K)/delta(theta_i)) - * - Compute beta^T * delta(K)/delta(theta_i) * beta - * 9: Update hyperparameters theta with Adam optimizer - * - m_T = beta1 * m_T-1 + (1 - beta1) * g_T - * - w_T = beta2 + w_T-1 + (1 - beta2) * g_T^2 - * - nu_T = nu * sqrt(1 - beta2_T) / (1 - beta1_T) - * - theta_T = theta_T-1 - nu_T * m_T / (sqrt(w_T) + epsilon) - */ - - // data holder for loss - hpx::shared_future loss_value; - - // Tiled future data structures - Tiled_matrix K_tiles; // Tiled covariance matrix K_NxN - Tiled_vector y_tiles; // Tiled output - Tiled_vector alpha_tiles; // Tiled intermediate solution - Tiled_matrix K_inv_tiles; // Tiled inversed covariance matrix K^-1_NxN - // Tiled future data structures for gradients - Tiled_matrix grad_v_tiles; // Tiled covariance with gradient v - Tiled_matrix grad_l_tiles; // Tiled covariance with gradient l - - // Preallocate memory - y_tiles.reserve(n_tiles); - - alpha_tiles.resize(n_tiles); // for now resize since reset in loop - K_inv_tiles.resize(n_tiles * n_tiles); // for now resize since reset in loop - - K_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure - grad_v_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure - grad_l_tiles.resize(n_tiles * n_tiles); // No reserve because of triangular structure - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous assembly of output y - for (std::size_t i = 0; i < n_tiles; i++) - { - y_tiles.push_back(detail::named_async("assemble_y", i, n_tile_size, training_output)); - } - - ////////////////////////////////////////////////////////////////////////////// - // Perform one optimization step - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous assembly of tiled covariance matrix, derivative of covariance matrix - // vector w.r.t. to vertical lengthscale and derivative of covariance - // matrix vector w.r.t. to lengthscale - for (std::size_t i = 0; i < n_tiles; i++) - { - for (std::size_t j = 0; j <= i; j++) - { - // Compute the distance (z_i - z_j) of K entries to reuse - auto cov_dists = detail::named_async( - "assemble_cov_dist", i, j, n_tile_size, n_regressors, sek_params, training_input); - - K_tiles[i * n_tiles + j] = detail::named_dataflow( - "assemble_K", i, j, n_tile_size, sek_params, cov_dists); - - if (trainable_params[0]) - { - grad_l_tiles[i * n_tiles + j] = - detail::named_dataflow("assemble_gradl", n_tile_size, sek_params, cov_dists); - if (i != j) - { - grad_l_tiles[j * n_tiles + i] = detail::named_dataflow( - "assemble_gradl_t", n_tile_size, n_tile_size, grad_l_tiles[i * n_tiles + j]); - } - } - - if (trainable_params[1]) - { - grad_v_tiles[i * n_tiles + j] = - detail::named_dataflow("assemble_gradv", n_tile_size, sek_params, cov_dists); - if (i != j) - { - grad_v_tiles[j * n_tiles + i] = detail::named_dataflow( - "assemble_gradv_t", n_tile_size, n_tile_size, grad_v_tiles[i * n_tiles + j]); - } - } - } - } - - // Assembly with reallocation -> optimize to only set existing values - for (std::size_t i = 0; i < n_tiles; i++) - { - alpha_tiles[i] = detail::named_async("assemble_tiled", n_tile_size); - } - - for (std::size_t i = 0; i < n_tiles; i++) - { - for (std::size_t j = 0; j < n_tiles; j++) - { - if (i == j) - { - K_inv_tiles[i * n_tiles + j] = - detail::named_async("assemble_identity_matrix", n_tile_size); - } - else - { - K_inv_tiles[i * n_tiles + j] = - detail::named_async("assemble_identity_matrix", n_tile_size * n_tile_size); - } - } - } - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(K_tiles, n_tile_size, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous compute K^-1 through L* (L^T * X) = I - forward_solve_tiled_matrix(K_tiles, K_inv_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); - backward_solve_tiled_matrix(K_tiles, K_inv_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous compute beta = inv(K) * y - matrix_vector_tiled(K_inv_tiles, y_tiles, alpha_tiles, n_tile_size, n_tile_size, n_tiles, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous loss computation where - // loss(theta) = 0.5 * ( log(det(K)) - y^T * K^-1 * y - N * log(2 * pi) ) - compute_loss_tiled(K_tiles, alpha_tiles, y_tiles, loss_value, n_tile_size, n_tiles); - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous update of the hyperparameters - if (trainable_params[0]) - { // lengthscale - update_hyperparameter_tiled( - K_inv_tiles, grad_l_tiles, alpha_tiles, adam_params, sek_params, n_tile_size, n_tiles, iter, 0); - } - if (trainable_params[1]) - { // vertical_lengthscale - update_hyperparameter_tiled( - K_inv_tiles, grad_v_tiles, alpha_tiles, adam_params, sek_params, n_tile_size, n_tiles, iter, 1); - } - if (trainable_params[2]) - { // noise_variance - update_hyperparameter_tiled( - K_inv_tiles, - Tiled_matrix{}, // no tiled gradient matrix required - alpha_tiles, - adam_params, - sek_params, - n_tile_size, - n_tiles, - iter, - 2); - } - return loss_value.get(); -} } // end of namespace cpu diff --git a/core/src/cpu/tiled_algorithms.cpp b/core/src/cpu/tiled_algorithms.cpp index 8989bb08..c2e4ba30 100644 --- a/core/src/cpu/tiled_algorithms.cpp +++ b/core/src/cpu/tiled_algorithms.cpp @@ -3,402 +3,29 @@ #include "gprat/cpu/adapter_cblas_fp64.hpp" #include "gprat/cpu/gp_algorithms.hpp" #include "gprat/cpu/gp_optimizer.hpp" -#include "gprat/cpu/gp_uncertainty.hpp" -#include "gprat/detail/async_helpers.hpp" - -#include GPRAT_NS_BEGIN namespace cpu { -// Tiled Cholesky Algorithm - -void right_looking_cholesky_tiled(Tiled_matrix &ft_tiles, std::size_t N, std::size_t n_tiles) -{ - for (std::size_t k = 0; k < n_tiles; k++) - { - // POTRF: Compute Cholesky factor L - ft_tiles[k * n_tiles + k] = detail::named_dataflow("cholesky_tiled", ft_tiles[k * n_tiles + k], N); - for (std::size_t m = k + 1; m < n_tiles; m++) - { - // TRSM: Solve X * L^T = A - ft_tiles[m * n_tiles + k] = detail::named_dataflow( - "cholesky_tiled", ft_tiles[k * n_tiles + k], ft_tiles[m * n_tiles + k], N, N, Blas_trans, Blas_right); - } - for (std::size_t m = k + 1; m < n_tiles; m++) - { - // SYRK: A = A - B * B^T - ft_tiles[m * n_tiles + m] = - detail::named_dataflow("cholesky_tiled", ft_tiles[m * n_tiles + m], ft_tiles[m * n_tiles + k], N); - for (std::size_t n = k + 1; n < m; n++) - { - // GEMM: C = C - A * B^T - ft_tiles[m * n_tiles + n] = detail::named_dataflow( - "cholesky_tiled", - ft_tiles[m * n_tiles + k], - ft_tiles[n * n_tiles + k], - ft_tiles[m * n_tiles + n], - N, - N, - N, - Blas_no_trans, - Blas_trans); - } - } - } -} - -// Tiled Triangular Solve Algorithms - -void forward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, std::size_t N, std::size_t n_tiles) -{ - for (std::size_t k = 0; k < n_tiles; k++) - { - // TRSM: Solve L * x = a - ft_rhs[k] = detail::named_dataflow( - "triangular_solve_tiled", ft_tiles[k * n_tiles + k], ft_rhs[k], N, Blas_no_trans); - for (std::size_t m = k + 1; m < n_tiles; m++) - { - // GEMV: b = b - A * a - ft_rhs[m] = detail::named_dataflow( - "triangular_solve_tiled", - ft_tiles[m * n_tiles + k], - ft_rhs[k], - ft_rhs[m], - N, - N, - Blas_substract, - Blas_no_trans); - } - } -} - -void backward_solve_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_rhs, std::size_t N, std::size_t n_tiles) -{ - for (int k_ = static_cast(n_tiles) - 1; k_ >= 0; k_--) // int instead of std::size_t for last comparison - { - std::size_t k = static_cast(k_); - // TRSM: Solve L^T * x = a - ft_rhs[k] = - detail::named_dataflow("triangular_solve_tiled", ft_tiles[k * n_tiles + k], ft_rhs[k], N, Blas_trans); - for (int m_ = k_ - 1; m_ >= 0; m_--) // int instead of std::size_t for last comparison - { - std::size_t m = static_cast(m_); - // GEMV:b = b - A^T * a - ft_rhs[m] = detail::named_dataflow( - "triangular_solve_tiled", - ft_tiles[k * n_tiles + m], - ft_rhs[k], - ft_rhs[m], - N, - N, - Blas_substract, - Blas_trans); - } - } -} - -void forward_solve_tiled_matrix(Tiled_matrix &ft_tiles, - Tiled_matrix &ft_rhs, - std::size_t N, - std::size_t M, - std::size_t n_tiles, - std::size_t m_tiles) -{ - for (std::size_t c = 0; c < m_tiles; c++) - { - for (std::size_t k = 0; k < n_tiles; k++) - { - // TRSM: solve L * X = A - ft_rhs[k * m_tiles + c] = detail::named_dataflow( - "triangular_solve_tiled_matrix", - ft_tiles[k * n_tiles + k], - ft_rhs[k * m_tiles + c], - N, - M, - Blas_no_trans, - Blas_left); - for (std::size_t m = k + 1; m < n_tiles; m++) - { - // GEMM: C = C - A * B - ft_rhs[m * m_tiles + c] = detail::named_dataflow( - "triangular_solve_tiled_matrix", - ft_tiles[m * n_tiles + k], - ft_rhs[k * m_tiles + c], - ft_rhs[m * m_tiles + c], - N, - M, - N, - Blas_no_trans, - Blas_no_trans); - } - } - } -} - -void backward_solve_tiled_matrix(Tiled_matrix &ft_tiles, - Tiled_matrix &ft_rhs, - std::size_t N, - std::size_t M, - std::size_t n_tiles, - std::size_t m_tiles) -{ - for (std::size_t c = 0; c < m_tiles; c++) - { - for (int k_ = static_cast(n_tiles) - 1; k_ >= 0; k_--) // int instead of std::size_t for last comparison - { - std::size_t k = static_cast(k_); - // TRSM: solve L^T * X = A - ft_rhs[k * m_tiles + c] = detail::named_dataflow( - "triangular_solve_tiled_matrix", - ft_tiles[k * n_tiles + k], - ft_rhs[k * m_tiles + c], - N, - M, - Blas_trans, - Blas_left); - for (int m_ = k_ - 1; m_ >= 0; m_--) // int instead of std::size_t for last comparison - { - std::size_t m = static_cast(m_); - // GEMM: C = C - A^T * B - ft_rhs[m * m_tiles + c] = detail::named_dataflow( - "triangular_solve_tiled_matrix", - ft_tiles[k * n_tiles + m], - ft_rhs[k * m_tiles + c], - ft_rhs[m * m_tiles + c], - N, - M, - N, - Blas_trans, - Blas_no_trans); - } - } - } -} - -void matrix_vector_tiled(Tiled_matrix &ft_tiles, - Tiled_vector &ft_vector, - Tiled_vector &ft_rhs, - std::size_t N_row, - std::size_t N_col, - std::size_t n_tiles, - std::size_t m_tiles) -{ - for (std::size_t k = 0; k < m_tiles; k++) - { - for (std::size_t m = 0; m < n_tiles; m++) - { - ft_rhs[k] = detail::named_dataflow( - "prediction_tiled", - ft_tiles[k * n_tiles + m], - ft_vector[m], - ft_rhs[k], - N_row, - N_col, - Blas_add, - Blas_no_trans); - } - } -} - -void symmetric_matrix_matrix_diagonal_tiled( - Tiled_matrix &ft_tiles, - Tiled_vector &ft_vector, - std::size_t N, - std::size_t M, - std::size_t n_tiles, - std::size_t m_tiles) -{ - for (std::size_t i = 0; i < m_tiles; ++i) - { - for (std::size_t n = 0; n < n_tiles; ++n) - { // Compute inner product to obtain diagonal elements of - // V^T * V <=> cross(K) * K^-1 * cross(K)^T - ft_vector[i] = - detail::named_dataflow("posterior_tiled", ft_tiles[n * m_tiles + i], ft_vector[i], N, M); - } - } -} - -void symmetric_matrix_matrix_tiled(Tiled_matrix &ft_tiles, - Tiled_matrix &ft_result, - std::size_t N, - std::size_t M, - std::size_t n_tiles, - std::size_t m_tiles) +namespace impl { - for (std::size_t c = 0; c < m_tiles; c++) - { - for (std::size_t k = 0; k < m_tiles; k++) - { - for (std::size_t m = 0; m < n_tiles; m++) - { - // (SYRK for (c == k) possible) - // GEMM: C = C - A^T * B - ft_result[c * m_tiles + k] = detail::named_dataflow( - "triangular_solve_tiled_matrix", - ft_tiles[m * m_tiles + c], - ft_tiles[m * m_tiles + k], - ft_result[c * m_tiles + k], - N, - M, - M, - Blas_trans, - Blas_no_trans); - } - } - } -} -void vector_difference_tiled(Tiled_vector &ft_minuend, Tiled_vector &ft_subtrahend, std::size_t M, std::size_t m_tiles) -{ - for (std::size_t i = 0; i < m_tiles; i++) - { - ft_subtrahend[i] = detail::named_dataflow("uncertainty_tiled", ft_minuend[i], ft_subtrahend[i], M); - } -} - -void matrix_diagonal_tiled(Tiled_matrix &ft_tiles, Tiled_vector &ft_vector, std::size_t M, std::size_t m_tiles) -{ - for (std::size_t i = 0; i < m_tiles; i++) - { - ft_vector[i] = detail::named_dataflow("uncertainty_tiled", ft_tiles[i * m_tiles + i], M); - } -} - -void compute_loss_tiled(Tiled_matrix &ft_tiles, - Tiled_vector &ft_alpha, - Tiled_vector &ft_y, - hpx::shared_future &loss, - std::size_t N, - std::size_t n_tiles) -{ - std::vector> loss_tiled; - loss_tiled.reserve(n_tiles); - for (std::size_t k = 0; k < n_tiles; k++) - { - loss_tiled.push_back( - detail::named_dataflow("loss_tiled", ft_tiles[k * n_tiles + k], ft_alpha[k], ft_y[k], N)); - } - - loss = detail::named_dataflow("loss_tiled", loss_tiled, N, n_tiles); -} - -void update_hyperparameter_tiled( - const Tiled_matrix &ft_invK, - const Tiled_matrix &ft_gradK_param, - const Tiled_vector &ft_alpha, +void update_parameters( const AdamParams &adam_params, SEKParams &sek_params, std::size_t N, std::size_t n_tiles, std::size_t iter, - std::size_t param_idx) + std::size_t param_idx, + double trace, + double dot, + bool jitter, + double factor) { - /* - * PART 1: - * Compute gradient = 0.5 * ( trace(inv(K) * grad(K)_param) + y^T * inv(K) * grad(K)_param * inv(K) * y ) - * - * 1: Compute trace(inv(K) * grad(K)_param) - * 2: Compute y^T * inv(K) * grad(K)_param * inv(K) * y - * - * Update parameter: - * 3: Update moments - * - m_T = beta1 * m_T-1 + (1 - beta1) * g_T - * - w_T = beta2 + w_T-1 + (1 - beta2) * g_T^2 - * 4: Adam step: - * - nu_T = nu * sqrt(1 - beta2_T) / (1 - beta1_T) - * - theta_T = theta_T-1 - nu_T * m_T / (sqrt(w_T) + epsilon) - */ - hpx::shared_future trace = hpx::make_ready_future(0.0); - hpx::shared_future dot = hpx::make_ready_future(0.0); - bool jitter = false; - double factor = 1.0; - if (param_idx == 0 || param_idx == 1) // 0: lengthscale; 1: vertical_lengthscale - { - Tiled_vector diag_tiles; // Diagonal tiles - Tiled_vector inter_alpha; // Intermediate result - // Preallocate memory - inter_alpha.reserve(n_tiles); - diag_tiles.reserve(n_tiles); - // Asynchrnonous initialization - for (std::size_t d = 0; d < n_tiles; d++) - { - diag_tiles.push_back(detail::named_async("assemble", N)); - inter_alpha.push_back(detail::named_async("assemble", N)); - } - - //////////////////////////////////// - // PART 1: Compute gradient - // Step 1: Compute trace(inv(K)*grad_K_param) - // Compute diagonal tiles of inv(K) * grad(K)_param - for (std::size_t i = 0; i < n_tiles; ++i) - { - for (std::size_t j = 0; j < n_tiles; ++j) - { - diag_tiles[i] = detail::named_dataflow( - "trace", ft_invK[i * n_tiles + j], ft_gradK_param[j * n_tiles + i], diag_tiles[i], N, N); - } - } - // Compute the trace of the diagonal tiles - for (std::size_t j = 0; j < n_tiles; ++j) - { - trace = detail::named_dataflow("trace", diag_tiles[j], trace); - } - // Not sure if can be done this way - // Step 2: Compute alpha^T * grad(K)_param * alpha (with alpha = inv(K) * y) - // Compute inter_alpha = grad(K)_param * alpha - for (std::size_t k = 0; k < n_tiles; k++) - { - for (std::size_t m = 0; m < n_tiles; m++) - { - inter_alpha[k] = detail::named_dataflow( - "gemv", - ft_gradK_param[k * n_tiles + m], - ft_alpha[m], - inter_alpha[k], - N, - N, - Blas_add, - Blas_no_trans); - } - } - // Compute alpha^T * inter_alpha - for (std::size_t j = 0; j < n_tiles; ++j) - { - dot = detail::named_dataflow("grad_right_tiled", inter_alpha[j], ft_alpha[j], dot); - } - } - else if (param_idx == 2) // @2: noise_variance - { - jitter = true; - //////////////////////////////////// - // PART 1: Compute gradient - // Step 1: Compute the trace of inv(K) * noise_variance - for (std::size_t j = 0; j < n_tiles; ++j) - { - trace = detail::named_dataflow("grad_left_tiled", ft_invK[j * n_tiles + j], trace, N); - } - //////////////////////////////////// - // Step 2: Compute the alpha^T * alpha * noise_variance - for (std::size_t j = 0; j < n_tiles; ++j) - { - dot = detail::named_dataflow("grad_right_tiled", ft_alpha[j], ft_alpha[j], dot); - } - - factor = compute_sigmoid(to_unconstrained(sek_params.noise_variance, true)); - } - else - { - // Throw an exception for invalid param_idx - throw std::invalid_argument("Invalid param_idx"); - } - // Compute gradient = trace + dot - double gradient = - factor * detail::named_dataflow("update_hyperparam", trace, dot, N, n_tiles).get(); + double gradient = factor * compute_gradient(trace, dot, N, n_tiles); //////////////////////////////////// // PART 2: Update parameter @@ -412,16 +39,14 @@ void update_hyperparameter_tiled( double unconstrained_param = to_unconstrained(sek_params.get_param(param_idx), jitter); // Adam step update with unconstrained parameter // compute beta_t inside - double updated_param = adam_step( - unconstrained_param, - adam_params, - sek_params.m_T[param_idx], - sek_params.w_T[param_idx], - static_cast(iter)); + double updated_param = + adam_step(unconstrained_param, adam_params, sek_params.m_T[param_idx], sek_params.w_T[param_idx], iter); // Transform hyperparameter back to constrained form sek_params.set_param(param_idx, to_constrained(updated_param, jitter)); } +} + } // end of namespace cpu GPRAT_NS_END diff --git a/core/src/gprat.cpp b/core/src/gprat.cpp index 858aa672..a843afb9 100644 --- a/core/src/gprat.cpp +++ b/core/src/gprat.cpp @@ -102,259 +102,184 @@ std::vector GP::get_training_output() const { return training_output_; } std::vector GP::predict(const std::vector &test_input, std::size_t m_tiles, std::size_t m_tile_size) { - return hpx::async( - [this, &test_input, m_tiles, m_tile_size]() - { #if GPRAT_WITH_CUDA - if (target_->is_gpu()) - { - return gpu::predict( - training_input_, - training_output_, - test_input, - kernel_params, - n_tiles_, - n_tile_size_, - m_tiles, - m_tile_size, - n_reg, - *std::dynamic_pointer_cast(target_)); - } - else - { - return cpu::predict( - training_input_, - training_output_, - test_input, - kernel_params, - n_tiles_, - n_tile_size_, - m_tiles, - m_tile_size, - n_reg); - } -#else - return cpu::predict( - training_input_, - training_output_, - test_input, - kernel_params, - n_tiles_, - n_tile_size_, - m_tiles, - m_tile_size, - n_reg); + if (target_->is_gpu()) + { + return gpu::predict( + training_input_, + training_output_, + test_input, + kernel_params, + n_tiles_, + n_tile_size_, + m_tiles, + m_tile_size, + n_reg, + *std::dynamic_pointer_cast(target_)); + } #endif - }) - .get(); + + tiled_scheduler_local scheduler; + return cpu::predict( + scheduler, + training_input_, + training_output_, + test_input, + kernel_params, + n_tiles_, + n_tile_size_, + m_tiles, + m_tile_size, + n_reg); } std::vector> GP::predict_with_uncertainty(const std::vector &test_input, std::size_t m_tiles, std::size_t m_tile_size) { - return hpx::async( - [this, &test_input, m_tiles, m_tile_size]() - { #if GPRAT_WITH_CUDA - if (target_->is_gpu()) - { - return gpu::predict_with_uncertainty( - training_input_, - training_output_, - test_input, - kernel_params, - n_tiles_, - n_tile_size_, - m_tiles, - m_tile_size, - n_reg, - *std::dynamic_pointer_cast(target_)); - } - else - { - return cpu::predict_with_uncertainty( - training_input_, - training_output_, - test_input, - kernel_params, - n_tiles_, - n_tile_size_, - m_tiles, - m_tile_size, - n_reg); - } -#else - return cpu::predict_with_uncertainty( - training_input_, - training_output_, - test_input, - kernel_params, - n_tiles_, - n_tile_size_, - m_tiles, - m_tile_size, - n_reg); + if (target_->is_gpu()) + { + return gpu::predict_with_uncertainty( + training_input_, + training_output_, + test_input, + kernel_params, + n_tiles_, + n_tile_size_, + m_tiles, + m_tile_size, + n_reg, + *std::dynamic_pointer_cast(target_)); + } #endif - }) - .get(); + tiled_scheduler_local scheduler; + return cpu::predict_with_uncertainty( + scheduler, + training_input_, + training_output_, + test_input, + kernel_params, + n_tiles_, + n_tile_size_, + m_tiles, + m_tile_size, + n_reg); } std::vector> GP::predict_with_full_cov(const std::vector &test_input, std::size_t m_tiles, std::size_t m_tile_size) { - return hpx::async( - [this, &test_input, m_tiles, m_tile_size]() - { #if GPRAT_WITH_CUDA - if (target_->is_gpu()) - { - return gpu::predict_with_full_cov( - training_input_, - training_output_, - test_input, - kernel_params, - n_tiles_, - n_tile_size_, - m_tiles, - m_tile_size, - n_reg, - *std::dynamic_pointer_cast(target_)); - } - else - { - return cpu::predict_with_full_cov( - training_input_, - training_output_, - test_input, - kernel_params, - n_tiles_, - n_tile_size_, - m_tiles, - m_tile_size, - n_reg); - } -#else - return cpu::predict_with_full_cov( - training_input_, - training_output_, - test_input, - kernel_params, - n_tiles_, - n_tile_size_, - m_tiles, - m_tile_size, - n_reg); + if (target_->is_gpu()) + { + return gpu::predict_with_full_cov( + training_input_, + training_output_, + test_input, + kernel_params, + n_tiles_, + n_tile_size_, + m_tiles, + m_tile_size, + n_reg, + *std::dynamic_pointer_cast(target_)); + } #endif - }) - .get(); + tiled_scheduler_local scheduler; + return cpu::predict_with_full_cov( + scheduler, + training_input_, + training_output_, + test_input, + kernel_params, + n_tiles_, + n_tile_size_, + m_tiles, + m_tile_size, + n_reg); } std::vector GP::optimize(const AdamParams &adam_params) { - return hpx::async( - [this, &adam_params]() - { #if GPRAT_WITH_CUDA - if (target_->is_gpu()) - { - std::cerr << "GP::optimze_step has not been implemented for the GPU.\n" - << "Instead, this operation executes the CPU implementation." << std::endl; - } + if (target_->is_gpu()) + { + std::cerr << "GP::optimze_step has not been implemented for the GPU.\n" + << "Instead, this operation executes the CPU implementation." << std::endl; + } #endif - return cpu::optimize( - training_input_, - training_output_, - n_tiles_, - n_tile_size_, - n_reg, - adam_params, - kernel_params, - trainable_params_); - }) - .get(); + tiled_scheduler_local scheduler; + return cpu::optimize( + scheduler, + training_input_, + training_output_, + n_tiles_, + n_tile_size_, + n_reg, + adam_params, + kernel_params, + trainable_params_); } double GP::optimize_step(AdamParams &adam_params, std::size_t iter) { - return hpx::async( - [this, &adam_params, iter]() - { #if GPRAT_WITH_CUDA - if (target_->is_gpu()) - { - std::cerr << "GP::optimze_step has not been implemented for the GPU.\n" - << "Instead, this operation executes the CPU implementation." << std::endl; - } + if (target_->is_gpu()) + { + std::cerr << "GP::optimze_step has not been implemented for the GPU.\n" + << "Instead, this operation executes the CPU implementation." << std::endl; + } #endif - return cpu::optimize_step( - training_input_, - training_output_, - n_tiles_, - n_tile_size_, - n_reg, - adam_params, - kernel_params, - trainable_params_, - iter); - }) - .get(); + tiled_scheduler_local scheduler; + return cpu::optimize_step( + scheduler, + training_input_, + training_output_, + n_tiles_, + n_tile_size_, + n_reg, + adam_params, + kernel_params, + trainable_params_, + iter); } double GP::calculate_loss() { - return hpx::async( - [this]() - { #if GPRAT_WITH_CUDA - if (target_->is_gpu()) - { - return gpu::compute_loss( - training_input_, - training_output_, - kernel_params, - n_tiles_, - n_tile_size_, - n_reg, - *std::dynamic_pointer_cast(target_)); - } - else - { - return cpu::compute_loss( - training_input_, training_output_, kernel_params, n_tiles_, n_tile_size_, n_reg); - } -#else - return cpu::compute_loss( - training_input_, training_output_, kernel_params, n_tiles_, n_tile_size_, n_reg); + if (target_->is_gpu()) + { + return gpu::compute_loss( + training_input_, + training_output_, + kernel_params, + n_tiles_, + n_tile_size_, + n_reg, + *std::dynamic_pointer_cast(target_)); + } #endif - }) - .get(); + tiled_scheduler_local scheduler; + return cpu::calculate_loss( + scheduler, training_input_, training_output_, kernel_params, n_tiles_, n_tile_size_, n_reg); } std::vector> GP::cholesky() { - return hpx::async( - [this]() - { #if GPRAT_WITH_CUDA - if (target_->is_gpu()) - { - return gpu::cholesky( - training_input_, - kernel_params, - n_tiles_, - n_tile_size_, - n_reg, - *std::dynamic_pointer_cast(target_)); - } - else - { - return cpu::cholesky(training_input_, kernel_params, n_tiles_, n_tile_size_, n_reg); - } -#else - return cpu::cholesky(training_input_, kernel_params, n_tiles_, n_tile_size_, n_reg); + if (target_->is_gpu()) + { + return gpu::cholesky( + training_input_, + kernel_params, + n_tiles_, + n_tile_size_, + n_reg, + *std::dynamic_pointer_cast(target_)); + } #endif - }) - .get(); + tiled_scheduler_local sched; + return cpu::cholesky(sched, training_input_, kernel_params, n_tiles_, n_tile_size_, n_reg); } GPRAT_NS_END From 48a8a766dfdd029554efa3672fe1251e69d9b434 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sun, 10 Aug 2025 21:29:56 +0200 Subject: [PATCH 20/56] chore: Upgrade dependencies This is required because newer CMake versions don't support cmake_minimum_required with minimum versions <= 3.5 --- bindings/CMakeLists.txt | 2 +- vcpkg.json | 14 ++++++++++++-- 2 files changed, 13 insertions(+), 3 deletions(-) diff --git a/bindings/CMakeLists.txt b/bindings/CMakeLists.txt index bad4b5ea..5ae3222f 100644 --- a/bindings/CMakeLists.txt +++ b/bindings/CMakeLists.txt @@ -1,5 +1,5 @@ # try finding pybind11 -set(GPRat_pybind11_VERSION 2.10.3) +set(GPRat_pybind11_VERSION 2.13.6) find_package(pybind11 ${GPRat_pybind11_VERSION} QUIET) if(pybind11_FOUND) message(STATUS "Found package pybind11.") diff --git a/vcpkg.json b/vcpkg.json index 438621a2..0b252332 100644 --- a/vcpkg.json +++ b/vcpkg.json @@ -13,9 +13,19 @@ "name": "fmt" }, { - "name": "hpx" + "name": "hpx", + "features": [ + "cuda", + "bzip2", + "mpi", + "snappy", + "zlib" + ] + }, + { + "name": "cuda" } ], "default-features": [], - "builtin-baseline": "e08b7bd89ae162f8579df2f8d39a1ae94107c8fd" + "builtin-baseline": "365f6444ab40ee87c73c947b475b3a267b3cb77c" } From bbf35bcb93de8e346e1b755af83079ac012b9677 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sun, 9 Nov 2025 22:54:19 +0100 Subject: [PATCH 21/56] chore(core): Fix issues with CUDA / nvcc under Windows --- core/CMakeLists.txt | 9 +- core/include/gprat/cpu/gp_functions.hpp | 12 +- core/include/gprat/cpu/tiled_algorithms.hpp | 6 +- core/include/gprat/gpu/cuda_utils.cuh | 5 +- core/include/gprat/gpu/gp_algorithms.cuh | 5 +- core/include/gprat/gpu/gp_functions.cuh | 4 +- core/include/gprat/hyperparameters.hpp | 2 +- core/include/gprat/kernels.hpp | 2 +- core/include/gprat/scheduler.hpp | 158 ++++++++++++++++---- core/src/cpu/gp_algorithms.cpp | 3 +- core/src/cpu/tiled_algorithms.cpp | 2 +- core/src/gprat.cpp | 2 +- core/src/gpu/gp_algorithms.cu | 7 +- core/src/gpu/gp_functions.cu | 5 +- core/src/gpu/gp_optimizer.cu | 4 +- core/src/target.cpp | 3 +- 16 files changed, 166 insertions(+), 63 deletions(-) diff --git a/core/CMakeLists.txt b/core/CMakeLists.txt index 38f472ea..c907de00 100644 --- a/core/CMakeLists.txt +++ b/core/CMakeLists.txt @@ -2,12 +2,11 @@ option(GPRAT_WITH_CUDA "Enable GPU support with CUDA, cuSolver, cuBLAS" OFF) if(GPRAT_WITH_CUDA) + set(CMAKE_CUDA_STANDARD 20) + set(CMAKE_CUDA_EXTENSIONS OFF) enable_language(CUDA) endif() -# Pass variable to C++ code -add_compile_definitions(GPRAT_WITH_CUDA=$) - set(SOURCE_FILES src/gprat.cpp src/utils.cpp @@ -57,7 +56,10 @@ target_sources(gprat_core PRIVATE ${header_files}) target_link_libraries(gprat_core PUBLIC HPX::hpx) if(GPRAT_WITH_CUDA) + find_package(CUDAToolkit MODULE REQUIRED) target_link_libraries(gprat_core PUBLIC CUDA::cusolver CUDA::cublas) + # Flag not working for CLANG CUDA + target_compile_features(gprat_core PUBLIC cuda_std_${CMAKE_CUDA_STANDARD}) endif() # Include directories @@ -75,6 +77,7 @@ else() target_link_libraries(gprat_core PUBLIC ${OpenBLAS_LIB}) endif() +target_compile_definitions(gprat_core PUBLIC GPRAT_WITH_CUDA=$) target_compile_features(gprat_core PUBLIC cxx_std_20) set_property(TARGET gprat_core PROPERTY POSITION_INDEPENDENT_CODE ON) diff --git a/core/include/gprat/cpu/gp_functions.hpp b/core/include/gprat/cpu/gp_functions.hpp index a7aadbc1..55a9e0e3 100644 --- a/core/include/gprat/cpu/gp_functions.hpp +++ b/core/include/gprat/cpu/gp_functions.hpp @@ -730,12 +730,12 @@ std::vector> predict_with_full_cov( */ template double calculate_loss(Scheduler &sched, - const std::vector &training_input, - const std::vector &training_output, - const SEKParams &sek_params, - std::size_t n_tiles, - std::size_t n_tile_size, - std::size_t n_regressors) + const std::vector &training_input, + const std::vector &training_output, + const SEKParams &sek_params, + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors) { /* * Negative log likelihood loss: diff --git a/core/include/gprat/cpu/tiled_algorithms.hpp b/core/include/gprat/cpu/tiled_algorithms.hpp index 5cec2db5..718e4d5b 100644 --- a/core/include/gprat/cpu/tiled_algorithms.hpp +++ b/core/include/gprat/cpu/tiled_algorithms.hpp @@ -633,13 +633,15 @@ void update_hyperparameter_tiled_noise_variance( // Step 1: Compute the trace of inv(K) * noise_variance for (std::size_t j = 0; j < n_tiles; ++j) { - trace = detail::named_dataflow(sched, schedule::K_inv_tile(sched, n_tiles, j, j), "grad_left_tiled", ft_invK[j * n_tiles + j], trace, N); + trace = detail::named_dataflow( + sched, schedule::K_inv_tile(sched, n_tiles, j, j), "grad_left_tiled", ft_invK[j * n_tiles + j], trace, N); } //////////////////////////////////// // Step 2: Compute the alpha^T * alpha * noise_variance for (std::size_t j = 0; j < n_tiles; ++j) { - dot = detail::named_dataflow(sched, schedule::alpha_tile(sched, n_tiles, j),"grad_right_tiled", ft_alpha[j], ft_alpha[j], dot); + dot = detail::named_dataflow( + sched, schedule::alpha_tile(sched, n_tiles, j), "grad_right_tiled", ft_alpha[j], ft_alpha[j], dot); } factor = compute_sigmoid(to_unconstrained(sek_params.noise_variance, true)); diff --git a/core/include/gprat/gpu/cuda_utils.cuh b/core/include/gprat/gpu/cuda_utils.cuh index 029b248c..128c6e22 100644 --- a/core/include/gprat/gpu/cuda_utils.cuh +++ b/core/include/gprat/gpu/cuda_utils.cuh @@ -17,6 +17,8 @@ GPRAT_NS_BEGIN #define BLOCK_SIZE 16 +using hpx::cuda::experimental::check_cuda_error; + /** * @brief Copies a vector from the host to the device using the next CUDA stream * of gpu. @@ -31,7 +33,6 @@ GPRAT_NS_BEGIN */ inline double *copy_to_device(const std::vector &h_vector, CUDA_GPU &gpu) { - using hpx::cuda::experimental::check_cuda_error; double *d_vector; check_cuda_error(cudaMalloc(&d_vector, h_vector.size() * sizeof(double))); cudaStream_t stream = gpu.next_stream(); @@ -46,7 +47,6 @@ inline double *copy_to_device(const std::vector &h_vector, CUDA_GPU &gpu */ inline cusolverDnHandle_t create_cusolver_handle() { - using hpx::cuda::experimental::check_cuda_error; cusolverDnHandle_t handle; cusolverDnCreate(&handle); return handle; @@ -66,7 +66,6 @@ inline void destroy(cusolverDnHandle_t handle) { cusolverDnDestroy(handle); } */ inline void free(std::vector> &vector) { - using hpx::cuda::experimental::check_cuda_error; for (auto &ptr : vector) { check_cuda_error(cudaFree(ptr.get())); diff --git a/core/include/gprat/gpu/gp_algorithms.cuh b/core/include/gprat/gpu/gp_algorithms.cuh index d78e1160..8da8a956 100644 --- a/core/include/gprat/gpu/gp_algorithms.cuh +++ b/core/include/gprat/gpu/gp_algorithms.cuh @@ -4,9 +4,10 @@ #pragma once #include "gprat/detail/config.hpp" - #include "gprat/kernels.hpp" #include "gprat/target.hpp" +#include "gprat/tile_data.hpp" + #include #include @@ -304,7 +305,7 @@ std::vector copy_tiled_vector_to_host_vector(std::vector> move_lower_tiled_matrix_to_host( +std::vector> move_lower_tiled_matrix_to_host( const std::vector> &d_tiles, const std::size_t n_tile_size, const std::size_t n_tiles, diff --git a/core/include/gprat/gpu/gp_functions.cuh b/core/include/gprat/gpu/gp_functions.cuh index 780485df..d8746d33 100644 --- a/core/include/gprat/gpu/gp_functions.cuh +++ b/core/include/gprat/gpu/gp_functions.cuh @@ -4,10 +4,10 @@ #pragma once #include "gprat/detail/config.hpp" - #include "gprat/hyperparameters.hpp" #include "gprat/kernels.hpp" #include "gprat/target.hpp" +#include "gprat/tile_data.hpp" GPRAT_NS_BEGIN @@ -192,7 +192,7 @@ double optimize_step(const std::vector &training_input, * * @return The tiled Cholesky factor */ -std::vector> +std::vector> cholesky(const std::vector &training_input, const SEKParams &sek_params, int n_tiles, diff --git a/core/include/gprat/hyperparameters.hpp b/core/include/gprat/hyperparameters.hpp index c980bd74..dae073dc 100644 --- a/core/include/gprat/hyperparameters.hpp +++ b/core/include/gprat/hyperparameters.hpp @@ -78,7 +78,7 @@ void load_construct_data(Archive &ar, AdamParams *v, const unsigned int) ar >> epsilon; ar >> opt_iter; - std::construct_at(v, learning_rate, beta1, beta2, epsilon, opt_iter); + new (v) AdamParams(learning_rate, beta1, beta2, epsilon, opt_iter); } GPRAT_NS_END diff --git a/core/include/gprat/kernels.hpp b/core/include/gprat/kernels.hpp index 0b489089..daa7798b 100644 --- a/core/include/gprat/kernels.hpp +++ b/core/include/gprat/kernels.hpp @@ -96,7 +96,7 @@ void load_construct_data(Archive &ar, SEKParams *v, const unsigned int) ar >> vertical_lengthscale; ar >> noise_variance; - std::construct_at(v, lengthscale, vertical_lengthscale, noise_variance); + new (v) SEKParams(lengthscale, vertical_lengthscale, noise_variance); } template diff --git a/core/include/gprat/scheduler.hpp b/core/include/gprat/scheduler.hpp index e19af509..2da7ccd7 100644 --- a/core/include/gprat/scheduler.hpp +++ b/core/include/gprat/scheduler.hpp @@ -7,6 +7,7 @@ // TODO: move to separate header #include "gprat/tile_data.hpp" + #include #include @@ -27,62 +28,155 @@ struct tile_dataset_type }; template -tiled_dataset_local -make_tiled_dataset(const tiled_scheduler_local &, std::size_t num_tiles, Mapper &&) +tiled_dataset_local make_tiled_dataset(const tiled_scheduler_local &, std::size_t num_tiles, Mapper &&) { return std::vector>>{ num_tiles }; } /// @brief This namespace contains the operation placement functions for all schedulers. -namespace schedule { +namespace schedule +{ #ifdef _MSC_VER #pragma warning(push) -#pragma warning(disable:4100) +#pragma warning(disable : 4100) #endif // ============================================================= // local scheduler -constexpr std::size_t covariance_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return 0; } -constexpr std::size_t cross_covariance_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return 0; } -constexpr std::size_t alpha_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t i) { return 0; } -constexpr std::size_t prediction_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t i) { return 0; } -constexpr std::size_t t_cross_covariance_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return 0; } -constexpr std::size_t prior_K_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return 0; } -constexpr std::size_t K_inv_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return 0; } -constexpr std::size_t K_grad_v_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return 0; } -constexpr std::size_t K_grad_l_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return 0; } -constexpr std::size_t uncertainty_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t i) { return 0; } -constexpr std::size_t inter_alpha_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t i) { return 0; } -constexpr std::size_t diag_tile(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t i) { return 0; } +constexpr std::size_t +covariance_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +{ + return 0; +} + +constexpr std::size_t +cross_covariance_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +{ + return 0; +} + +constexpr std::size_t alpha_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t i) { return 0; } + +constexpr std::size_t prediction_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t i) +{ + return 0; +} + +constexpr std::size_t +t_cross_covariance_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +{ + return 0; +} + +constexpr std::size_t +prior_K_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +{ + return 0; +} + +constexpr std::size_t +K_inv_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +{ + return 0; +} -constexpr std::size_t cholesky_potrf(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k) { return 0; } -constexpr std::size_t cholesky_syrk(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t m) { return 0; } -constexpr std::size_t cholesky_trsm(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k, std::size_t m) { return 0; } -constexpr std::size_t cholesky_gemm(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k, std::size_t m, std::size_t n) { return 0; } +constexpr std::size_t +K_grad_v_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +{ + return 0; +} -constexpr std::size_t solve_trsv(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k) { return 0; } -constexpr std::size_t solve_trsm(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k) { return 0; } -constexpr std::size_t solve_gemv(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k, std::size_t m) { return 0; } +constexpr std::size_t +K_grad_l_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +{ + return 0; +} -constexpr std::size_t solve_matrix_trsm(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t c,std::size_t k) { return 0; } -constexpr std::size_t solve_matrix_gemm(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t c,std::size_t k, std::size_t m) { return 0; } +constexpr std::size_t uncertainty_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t i) +{ + return 0; +} -constexpr std::size_t multiply_gemv(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k, std::size_t m) { return 0; } +constexpr std::size_t inter_alpha_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t i) +{ + return 0; +} -constexpr std::size_t k_rank_dot_diag_syrk(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k) { return 0; } -constexpr std::size_t k_rank_gemm(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t c,std::size_t k, std::size_t m) { return 0; } +constexpr std::size_t diag_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t i) { return 0; } -constexpr std::size_t vector_axpy(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k) { return 0; } -constexpr std::size_t get_diagonal(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k) { return 0; } -constexpr std::size_t compute_loss(const tiled_scheduler_local& sched, std::size_t n_tiles, std::size_t k) { return 0; } +constexpr std::size_t cholesky_potrf(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k) +{ + return 0; +} + +constexpr std::size_t cholesky_syrk(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t m) +{ + return 0; +} + +constexpr std::size_t +cholesky_trsm(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k, std::size_t m) +{ + return 0; +} + +constexpr std::size_t +cholesky_gemm(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k, std::size_t m, std::size_t n) +{ + return 0; +} + +constexpr std::size_t solve_trsv(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k) { return 0; } + +constexpr std::size_t solve_trsm(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k) { return 0; } + +constexpr std::size_t solve_gemv(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k, std::size_t m) +{ + return 0; +} + +constexpr std::size_t +solve_matrix_trsm(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t c, std::size_t k) +{ + return 0; +} + +constexpr std::size_t +solve_matrix_gemm(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) +{ + return 0; +} + +constexpr std::size_t +multiply_gemv(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k, std::size_t m) +{ + return 0; +} + +constexpr std::size_t k_rank_dot_diag_syrk(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k) +{ + return 0; +} + +constexpr std::size_t +k_rank_gemm(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) +{ + return 0; +} + +constexpr std::size_t vector_axpy(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k) { return 0; } + +constexpr std::size_t get_diagonal(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k) { return 0; } + +constexpr std::size_t compute_loss(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k) { return 0; } #ifdef _MSC_VER #pragma warning(pop) #endif -} +} // namespace schedule GPRAT_NS_END diff --git a/core/src/cpu/gp_algorithms.cpp b/core/src/cpu/gp_algorithms.cpp index 49f883dc..ab3ed77b 100644 --- a/core/src/cpu/gp_algorithms.cpp +++ b/core/src/cpu/gp_algorithms.cpp @@ -1,6 +1,7 @@ #include "gprat/cpu/gp_algorithms.hpp" -#include "gprat/tile_data.hpp" + #include "gprat/performance_counters.hpp" +#include "gprat/tile_data.hpp" #include diff --git a/core/src/cpu/tiled_algorithms.cpp b/core/src/cpu/tiled_algorithms.cpp index c2e4ba30..d035b89d 100644 --- a/core/src/cpu/tiled_algorithms.cpp +++ b/core/src/cpu/tiled_algorithms.cpp @@ -45,7 +45,7 @@ void update_parameters( sek_params.set_param(param_idx, to_constrained(updated_param, jitter)); } -} +} // namespace impl } // end of namespace cpu diff --git a/core/src/gprat.cpp b/core/src/gprat.cpp index a843afb9..969fdb9e 100644 --- a/core/src/gprat.cpp +++ b/core/src/gprat.cpp @@ -4,7 +4,7 @@ #include "gprat/utils.hpp" #if GPRAT_WITH_CUDA -#include "gpu/gp_functions.cuh" +#include "gprat/gpu/gp_functions.cuh" #endif GPRAT_NS_BEGIN diff --git a/core/src/gpu/gp_algorithms.cu b/core/src/gpu/gp_algorithms.cu index 5e80df22..b9125e57 100644 --- a/core/src/gpu/gp_algorithms.cu +++ b/core/src/gpu/gp_algorithms.cu @@ -5,6 +5,7 @@ #include "gprat/gpu/gp_optimizer.cuh" #include "gprat/kernels.hpp" #include "gprat/target.hpp" +#include "gprat/tile_data.hpp" #include #include @@ -533,13 +534,13 @@ std::vector copy_tiled_vector_to_host_vector( return h_vector; } -std::vector> move_lower_tiled_matrix_to_host( +std::vector> move_lower_tiled_matrix_to_host( const std::vector> &d_tiles, const std::size_t n_tile_size, const std::size_t n_tiles, CUDA_GPU &gpu) { - std::vector> h_tiles(n_tiles * n_tiles); + std::vector> h_tiles(n_tiles * n_tiles); std::vector streams(n_tiles * (n_tiles + 1) / 2); for (std::size_t i = 0; i < n_tiles; ++i) @@ -547,7 +548,7 @@ std::vector> move_lower_tiled_matrix_to_host( for (std::size_t j = 0; j <= i; ++j) { streams[i] = gpu.next_stream(); - h_tiles[i * n_tiles + j].resize(n_tile_size * n_tile_size); + h_tiles[i * n_tiles + j] = mutable_tile_data(n_tile_size * n_tile_size); check_cuda_error(cudaMemcpyAsync( h_tiles[i * n_tiles + j].data(), d_tiles[i * n_tiles + j].get(), diff --git a/core/src/gpu/gp_functions.cu b/core/src/gpu/gp_functions.cu index 80d40763..a4485992 100644 --- a/core/src/gpu/gp_functions.cu +++ b/core/src/gpu/gp_functions.cu @@ -5,6 +5,7 @@ #include "gprat/gpu/tiled_algorithms.cuh" #include "gprat/kernels.hpp" #include "gprat/target.hpp" +#include "gprat/tile_data.hpp" #include #include @@ -306,7 +307,7 @@ double optimize_step(const std::vector &training_input, // return 0.0; } -std::vector> +std::vector> cholesky(const std::vector &h_training_input, const SEKParams &sek_params, int n_tiles, @@ -326,7 +327,7 @@ cholesky(const std::vector &h_training_input, right_looking_cholesky_tiled(d_tiles, n_tile_size, n_tiles, gpu, cusolver); // Copy tiled matrix to host - std::vector> h_tiles = move_lower_tiled_matrix_to_host(d_tiles, n_tile_size, n_tiles, gpu); + auto h_tiles = move_lower_tiled_matrix_to_host(d_tiles, n_tile_size, n_tiles, gpu); cudaFree(d_training_input); destroy(cusolver); diff --git a/core/src/gpu/gp_optimizer.cu b/core/src/gpu/gp_optimizer.cu index 62727414..ea465261 100644 --- a/core/src/gpu/gp_optimizer.cu +++ b/core/src/gpu/gp_optimizer.cu @@ -4,6 +4,8 @@ #include "gprat/gpu/cuda_kernels.cuh" #include "gprat/gpu/cuda_utils.cuh" +#include + GPRAT_NS_BEGIN namespace gpu @@ -235,7 +237,7 @@ add_losses(const std::vector> &losses, std::size_t n_ { l += losses[i].get(); } - l += n_tile_size * n_tiles * log(2.0 * M_PI); + l += n_tile_size * n_tiles * log(2.0 * std::numbers::pi); return hpx::make_ready_future(0.5 * l / (n_tile_size * n_tiles)); } diff --git a/core/src/target.cpp b/core/src/target.cpp index 6b04618f..3cd90504 100644 --- a/core/src/target.cpp +++ b/core/src/target.cpp @@ -3,8 +3,7 @@ #include #if GPRAT_WITH_CUDA -#include "gpu/cuda_utils.cuh" -using hpx::cuda::experimental::check_cuda_error; +#include "gprat/gpu/cuda_utils.cuh" #endif GPRAT_NS_BEGIN From ae13f9d1984fc362a216dfa7b2d053b1335b0d9d Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sun, 9 Nov 2025 22:57:50 +0100 Subject: [PATCH 22/56] feat(core): Add optional cache eviction before BLAS operation --- CMakeLists.txt | 2 ++ core/CMakeLists.txt | 7 +++++- core/include/gprat/performance_counters.hpp | 19 ++++++++++++++ core/include/gprat/tile_data.hpp | 6 +++-- core/src/cpu/adapter_cblas_fp32.cpp | 22 ++++++++++++++++ core/src/cpu/adapter_cblas_fp64.cpp | 22 ++++++++++++++++ core/src/performance_counters.cpp | 28 +++++++++++++++++++++ 7 files changed, 103 insertions(+), 3 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 794a66f9..f6d7cf40 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -20,6 +20,8 @@ cmake_dependent_option(GPRAT_ENABLE_TESTS "Build unit and integration tests" ${PROJECT_IS_TOP_LEVEL} "GPRAT_BUILD_CORE" OFF) cmake_dependent_option(GPRAT_ENABLE_MKL "Enable support for Intel oneMKL" ${PROJECT_IS_TOP_LEVEL} "GPRAT_BUILD_CORE" OFF) +option(GPRAT_ENABLE_BENCHMARK_CACHE_EVICTIONS + "Evict data from caches before running BLAS operations" ON) option(GPRAT_ENABLE_FORMAT_TARGETS "Enable clang-format / cmake-format targets" ${PROJECT_IS_TOP_LEVEL}) diff --git a/core/CMakeLists.txt b/core/CMakeLists.txt index c907de00..1a7b4db3 100644 --- a/core/CMakeLists.txt +++ b/core/CMakeLists.txt @@ -77,7 +77,12 @@ else() target_link_libraries(gprat_core PUBLIC ${OpenBLAS_LIB}) endif() -target_compile_definitions(gprat_core PUBLIC GPRAT_WITH_CUDA=$) +target_compile_definitions(gprat_core + PUBLIC GPRAT_WITH_CUDA=$) +if(GPRAT_ENABLE_BENCHMARK_CACHE_EVICTIONS) + target_compile_definitions(gprat_core + PUBLIC GPRAT_ENABLE_BENCHMARK_CACHE_EVICTIONS) +endif() target_compile_features(gprat_core PUBLIC cxx_std_20) set_property(TARGET gprat_core PROPERTY POSITION_INDEPENDENT_CODE ON) diff --git a/core/include/gprat/performance_counters.hpp b/core/include/gprat/performance_counters.hpp index e347faff..13054735 100644 --- a/core/include/gprat/performance_counters.hpp +++ b/core/include/gprat/performance_counters.hpp @@ -11,6 +11,7 @@ #include #include #include +#include GPRAT_NS_BEGIN @@ -78,6 +79,24 @@ void track_tile_data_deallocation(std::size_t size); void register_performance_counters(); +void force_evict_memory(const void *start, std::size_t size); + +template +void force_evict_memory(std::span data) +{ + force_evict_memory(data.data(), data.size_bytes()); +} + +#ifdef GPRAT_ENABLE_BENCHMARK_CACHE_EVICTIONS +/// @brief Force-evict a memory span from the cache for benchmarking purposes. +/// @param data The memory region to evict +#define GPRAT_BENCHMARK_FORCE_EVICT(data) force_evict_memory(data) +#else +/// @brief Force-evict a memory span from the cache for benchmarking purposes. +/// @param data The memory region to evict +#define GPRAT_BENCHMARK_FORCE_EVICT(data) (void) data +#endif + GPRAT_NS_END #endif diff --git a/core/include/gprat/tile_data.hpp b/core/include/gprat/tile_data.hpp index 006ac62b..a2615ad8 100644 --- a/core/include/gprat/tile_data.hpp +++ b/core/include/gprat/tile_data.hpp @@ -112,6 +112,8 @@ class const_tile_data [[nodiscard]] const T &operator[](std::size_t idx) const { return cpu_data_[idx]; } + [[nodiscard]] std::span as_span() const noexcept { return { cpu_data_.data(), cpu_data_.size() }; } + // ReSharper disable once CppNonExplicitConversionOperator operator std::span() const noexcept // NOLINT(*-explicit-constructor) { @@ -157,10 +159,10 @@ class mutable_tile_data : public const_tile_data [[nodiscard]] T &operator[](std::size_t idx) const { return this->cpu_data_[idx]; } // ReSharper disable once CppNonExplicitConversionOperator - operator std::span() noexcept + operator std::span() noexcept // NOLINT(*-explicit-constructor) { return { this->cpu_data_.data(), this->cpu_data_.size() }; - } // NOLINT(*-explicit-constructor) + } }; GPRAT_NS_END diff --git a/core/src/cpu/adapter_cblas_fp32.cpp b/core/src/cpu/adapter_cblas_fp32.cpp index ca01a091..4cfbea51 100644 --- a/core/src/cpu/adapter_cblas_fp32.cpp +++ b/core/src/cpu/adapter_cblas_fp32.cpp @@ -21,6 +21,7 @@ GPRAT_NS_BEGIN mutable_tile_data potrf(const mutable_tile_data &A, const int N) { + GPRAT_BENCHMARK_FORCE_EVICT(A.as_span()); GPRAT_TIME_FUNCTION(&potrf); // POTRF: in-place Cholesky decomposition of A // use spotrf2 recursive version for better stability @@ -37,6 +38,8 @@ trsm(const const_tile_data &L, const BLAS_TRANSPOSE transpose_L, const BLAS_SIDE side_L) { + GPRAT_BENCHMARK_FORCE_EVICT(L.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(A.as_span()); GPRAT_TIME_FUNCTION(&trsm); // TRSM constants const float alpha = 1.0; @@ -59,6 +62,8 @@ trsm(const const_tile_data &L, mutable_tile_data syrk(const mutable_tile_data &A, const const_tile_data &B, const int N) { + GPRAT_BENCHMARK_FORCE_EVICT(A.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(B.as_span()); GPRAT_TIME_FUNCTION(&syrk); // SYRK constants const float alpha = -1.0; @@ -79,6 +84,9 @@ gemm(const const_tile_data &A, const BLAS_TRANSPOSE transpose_A, const BLAS_TRANSPOSE transpose_B) { + GPRAT_BENCHMARK_FORCE_EVICT(A.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(B.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(C.as_span()); GPRAT_TIME_FUNCTION(&gemm); // GEMM constants const float alpha = -1.0; @@ -108,6 +116,8 @@ gemm(const const_tile_data &A, mutable_tile_data trsv(const const_tile_data &L, const mutable_tile_data &a, const int N, const BLAS_TRANSPOSE transpose_L) { + GPRAT_BENCHMARK_FORCE_EVICT(L.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(a.as_span()); GPRAT_TIME_FUNCTION(&trsv); // TRSV: In-place solve L(^T) * x = a where L lower triangular cblas_strsv(CblasRowMajor, @@ -132,6 +142,9 @@ gemv(const const_tile_data &A, const BLAS_ALPHA alpha, const BLAS_TRANSPOSE transpose_A) { + GPRAT_BENCHMARK_FORCE_EVICT(A.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(a.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(b.as_span()); GPRAT_TIME_FUNCTION(&gemv); // GEMV constants // const float alpha = -1.0; @@ -157,6 +170,8 @@ gemv(const const_tile_data &A, mutable_tile_data dot_diag_syrk(const const_tile_data &A, const mutable_tile_data &r, const int N, const int M) { + GPRAT_BENCHMARK_FORCE_EVICT(A.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(r.as_span()); GPRAT_TIME_FUNCTION(&dot_diag_syrk); auto r_p = r.data(); auto A_p = A.data(); @@ -176,6 +191,9 @@ dot_diag_gemm(const const_tile_data &A, const int N, const int M) { + GPRAT_BENCHMARK_FORCE_EVICT(A.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(B.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(r.as_span()); GPRAT_TIME_FUNCTION(&dot_diag_gemm); auto r_p = r.data(); auto A_p = A.data(); @@ -192,6 +210,8 @@ dot_diag_gemm(const const_tile_data &A, mutable_tile_data axpy(const mutable_tile_data &y, const const_tile_data &x, const int N) { + GPRAT_BENCHMARK_FORCE_EVICT(y.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(x.as_span()); GPRAT_TIME_FUNCTION(&axpy); cblas_saxpy(N, -1.0, x.data(), 1, y.data(), 1); return y; @@ -199,6 +219,8 @@ mutable_tile_data axpy(const mutable_tile_data &y, const const_til float dot(std::span a, std::span b, const int N) { + GPRAT_BENCHMARK_FORCE_EVICT(a); + GPRAT_BENCHMARK_FORCE_EVICT(b); GPRAT_TIME_FUNCTION(&dot); // DOT: a * b return cblas_sdot(N, a.data(), 1, b.data(), 1); diff --git a/core/src/cpu/adapter_cblas_fp64.cpp b/core/src/cpu/adapter_cblas_fp64.cpp index f2e8b927..64c94c78 100644 --- a/core/src/cpu/adapter_cblas_fp64.cpp +++ b/core/src/cpu/adapter_cblas_fp64.cpp @@ -21,6 +21,7 @@ GPRAT_NS_BEGIN mutable_tile_data potrf(const mutable_tile_data &A, const int N) { + GPRAT_BENCHMARK_FORCE_EVICT(A.as_span()); GPRAT_TIME_FUNCTION(&potrf); // POTRF: in-place Cholesky decomposition of A // use dpotrf2 recursive version for better stability @@ -37,6 +38,8 @@ trsm(const const_tile_data &L, const BLAS_TRANSPOSE transpose_L, const BLAS_SIDE side_L) { + GPRAT_BENCHMARK_FORCE_EVICT(L.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(A.as_span()); GPRAT_TIME_FUNCTION(&trsm); // TRSM constants const double alpha = 1.0; @@ -60,6 +63,8 @@ trsm(const const_tile_data &L, mutable_tile_data syrk(const mutable_tile_data &A, const const_tile_data &B, const int N) { + GPRAT_BENCHMARK_FORCE_EVICT(A.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(B.as_span()); GPRAT_TIME_FUNCTION(&syrk); // SYRK constants const double alpha = -1.0; @@ -80,6 +85,9 @@ gemm(const const_tile_data &A, const BLAS_TRANSPOSE transpose_A, const BLAS_TRANSPOSE transpose_B) { + GPRAT_BENCHMARK_FORCE_EVICT(A.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(B.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(C.as_span()); GPRAT_TIME_FUNCTION(&gemm); // GEMM constants const double alpha = -1.0; @@ -109,6 +117,8 @@ gemm(const const_tile_data &A, mutable_tile_data trsv( const const_tile_data &L, const mutable_tile_data &a, const int N, const BLAS_TRANSPOSE transpose_L) { + GPRAT_BENCHMARK_FORCE_EVICT(L.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(a.as_span()); GPRAT_TIME_FUNCTION(&trsv); // TRSV: In-place solve L(^T) * x = a where L lower triangular cblas_dtrsv(CblasRowMajor, @@ -133,6 +143,9 @@ gemv(const const_tile_data &A, const BLAS_ALPHA alpha, const BLAS_TRANSPOSE transpose_A) { + GPRAT_BENCHMARK_FORCE_EVICT(A.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(a.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(b.as_span()); GPRAT_TIME_FUNCTION(&gemv); // GEMV constants // const double alpha = -1.0; @@ -158,6 +171,8 @@ gemv(const const_tile_data &A, mutable_tile_data dot_diag_syrk(const const_tile_data &A, const mutable_tile_data &r, const int N, const int M) { + GPRAT_BENCHMARK_FORCE_EVICT(A.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(r.as_span()); GPRAT_TIME_FUNCTION(&dot_diag_syrk); auto r_p = r.data(); auto A_p = A.data(); @@ -177,6 +192,9 @@ dot_diag_gemm(const const_tile_data &A, const int N, const int M) { + GPRAT_BENCHMARK_FORCE_EVICT(A.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(B.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(r.as_span()); GPRAT_TIME_FUNCTION(&dot_diag_gemm); auto r_p = r.data(); auto A_p = A.data(); @@ -193,6 +211,8 @@ dot_diag_gemm(const const_tile_data &A, mutable_tile_data axpy(const mutable_tile_data &y, const const_tile_data &x, const int N) { + GPRAT_BENCHMARK_FORCE_EVICT(y.as_span()); + GPRAT_BENCHMARK_FORCE_EVICT(x.as_span()); GPRAT_TIME_FUNCTION(&axpy); cblas_daxpy(N, -1.0, x.data(), 1, y.data(), 1); return y; @@ -200,6 +220,8 @@ mutable_tile_data axpy(const mutable_tile_data &y, const const_t double dot(std::span a, std::span b, const int N) { + GPRAT_BENCHMARK_FORCE_EVICT(a); + GPRAT_BENCHMARK_FORCE_EVICT(b); GPRAT_TIME_FUNCTION(&dot); // DOT: a * b return cblas_ddot(N, a.data(), 1, b.data(), 1); diff --git a/core/src/performance_counters.cpp b/core/src/performance_counters.cpp index c405cff4..0434e2bb 100644 --- a/core/src/performance_counters.cpp +++ b/core/src/performance_counters.cpp @@ -1,6 +1,7 @@ #include "gprat/performance_counters.hpp" #include +#include #include #ifdef HPX_HAVE_MODULE_PERFORMANCE_COUNTERS #include @@ -48,6 +49,7 @@ void register_performance_counters() detail::register_fp32_performance_counters(); detail::register_fp64_performance_counters(); } + #else void register_performance_counters() { @@ -55,4 +57,30 @@ void register_performance_counters() } #endif +void force_evict_memory(const void *start, std::size_t size) +{ + // A cache line size of 64 seems to be a safe estimate. + // see: https://lemire.me/blog/2023/12/12/measuring-the-size-of-the-cache-line-empirically/ + constexpr std::size_t cache_line_size = 64; + + const char *p = static_cast(start); + const char *end = p + size; + + _mm_mfence(); + do { + // Intel recommends clflushopt over normal clflush due to higher performance, see: + // http://www.intel.com/content/dam/www/public/us/en/documents/manuals/64-ia-32-architectures-optimization-manual.pdf + _mm_clflush(p); + p += cache_line_size; + } while (p < end); + + // Make sure we don't miss a cache line at the end + if ((reinterpret_cast(p) & (cache_line_size - 1)) + != (reinterpret_cast(end - 1) & (cache_line_size - 1))) + { + _mm_clflush(end - 1); + } + _mm_mfence(); +} + GPRAT_NS_END From 068da77ca41fe9ff30a79b73b46517a589f2bea5 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sun, 23 Nov 2025 01:16:32 +0100 Subject: [PATCH 23/56] chore: Add some minimal docs on Windows support --- README.md | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index 389bc776..7c73f0e0 100644 --- a/README.md +++ b/README.md @@ -9,7 +9,7 @@ code. ## Dependencies -GPRat depends on [HPX](https://hpx-docs.stellar-group.org/latest/html/index.html) for asynchronous task-based parallelization. +GPRat depends on [HPX](https://hpx-docs.stellar-group.org/latest/html/index.html) for asynchronous task-based parallelization. Furthermore, for CPU-only BLAS computation GPRat requires [OpenBLAS](http://www.openmathlib.org/OpenBLAS/) or [MKL](https://www.intel.com/content/www/us/en/developer/tools/oneapi/onemkl.html). A [CUDA](https://developer.nvidia.com/cuda-toolkit) installation is required for GPU-only BLAS computations. @@ -20,6 +20,9 @@ A script to install and setup spack for `GPRat` is provided in [`spack-repo`](sp Spack environment configurations and setup scripts for CPU and GPU use are provided in [`spack-repo/environments`](spack-repo/environments). +Since Spack is not available on Windows, we also support dependency installation using vcpkg. +For now, vcpkg builds are only tested on Windows. + ## How To Compile GPRat makes use of [CMake presets][1] to simplify the process of configuring the project. @@ -35,6 +38,7 @@ ctest --preset=dev-linux As a developer, you may create a `CMakeUserPresets.json` file at the root of the project that contains additional presets local to your machine. In addition to the build configuration `dev-linux`, there are `release-linux`, `dev-linux-gpu`, and `release-linux-gpu`. +For Windows, we have similar presets called `dev-windows` and `release-windows`. The configurations suffixed with `-gpu` build the library with CUDA. GPRat can be build with or without Python bindings. From 25ad34055fa40da663517ec531290e0cff91317f Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Tue, 6 May 2025 23:32:42 +0200 Subject: [PATCH 24/56] feat(core): Add basic heuristic for tile count Based on our shared-memory experiments. --- core/include/gprat/utils.hpp | 10 ++++++++++ core/src/utils.cpp | 25 +++++++++++++++++++++++++ 2 files changed, 35 insertions(+) diff --git a/core/include/gprat/utils.hpp b/core/include/gprat/utils.hpp index 86a4ddd2..a137ebda 100644 --- a/core/include/gprat/utils.hpp +++ b/core/include/gprat/utils.hpp @@ -44,6 +44,16 @@ std::size_t compute_train_tile_size(std::size_t n_samples, std::size_t n_tiles); std::pair compute_test_tiles(std::size_t n_test, std::size_t n_tiles, std::size_t n_tile_size); +/** + * @brief Computes a good-enough guess for the number of tiles per dimension. + * + * This guess is based on experiments ran on a single dual-socket 64 core machine. + * It might not be appropriate for distributed scenarios. + * + * @param n Number of samples + */ +std::size_t guess_good_tile_count_per_dimension(std::size_t n); + /** * @brief Load data from file * diff --git a/core/src/utils.cpp b/core/src/utils.cpp index 47935bfd..6ebcb705 100644 --- a/core/src/utils.cpp +++ b/core/src/utils.cpp @@ -50,6 +50,31 @@ std::pair compute_test_tiles(std::size_t n_test, std:: return { m_tiles, m_tile_size }; } +std::size_t guess_good_tile_count_per_dimension(std::size_t n) +{ + // These have been found through experimentation - they are only estimates that have been shown to perform + // better than fixed tile counts. + const auto hw_concurrency = hpx::threads::hardware_concurrency(); + + // For small datasets / few cores we shouldn't bother + if (n < (1 << 8) || hw_concurrency < 4) + { + return 1; + } + + if (n < (1 << 12) || hw_concurrency < 16) + { + return 4; + } + + if (n < (1 << 18) || hw_concurrency < 32) + { + return 16; + } + + return std::min(hw_concurrency, n / 256); +} + std::vector load_data(const std::string &file_path, std::size_t n_samples, std::size_t offset) { std::vector _data; From cea46d911f56cd850ae80eb8d222911b0f66e85b Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Wed, 28 May 2025 04:35:54 +0200 Subject: [PATCH 25/56] fix(spack-repo): Enable HPX networking and instrumentation --- spack-repo/environments/spack_cpu_gcc.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/spack-repo/environments/spack_cpu_gcc.yaml b/spack-repo/environments/spack_cpu_gcc.yaml index e442106d..5b48f640 100644 --- a/spack-repo/environments/spack_cpu_gcc.yaml +++ b/spack-repo/environments/spack_cpu_gcc.yaml @@ -1,7 +1,7 @@ spack: specs: - - hpx@1.10.0%gcc +static malloc=system networking=none max_cpu_count=256 instrumentation=none ^cmake@3.30 ^curl@8.10.1 ^ninja@1.12.1 - - intel-oneapi-mkl@2024.2.1%gcc shared=false + - hpx@1.10.0 +static malloc=system max_cpu_count=256%gcc ^cmake ^curl ^ninja + - intel-oneapi-mkl@2024.2.1 shared=false - openblas@0.3.28 shared=false fortran=false view: true concretizer: From 9d8b1798dbe00cb0dedfb228628c9966ec9efa7b Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Wed, 28 May 2025 04:36:18 +0200 Subject: [PATCH 26/56] fix(spack-repo): Blacklist asio 1.34 for HPX 1.10 --- spack-repo/packages/hpx/package.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/spack-repo/packages/hpx/package.py b/spack-repo/packages/hpx/package.py index 0ed413b9..1f9e5cc3 100644 --- a/spack-repo/packages/hpx/package.py +++ b/spack-repo/packages/hpx/package.py @@ -171,6 +171,10 @@ class Hpx(CMakePackage, CudaPackage, ROCmPackage): # Patches and one-off conflicts + # Asio 1.34.0 removed io_context::work, used by HPX: + # https://github.com/chriskohlhoff/asio/commit/a70f2df321ff40c1809773c2c09986745abf8d20. + conflicts("^asio@1.34:", when="@:1.10") + # Certain Asio headers don't compile with nvcc from 1.17.0 onwards with # C++17. Starting with CUDA 11.3 they compile again. conflicts("^asio@1.17.0:", when="+cuda cxxstd=17 ^cuda@:11.2") From bfbfcc9a7372bd2c1d65089dd2555c6e5c8e6017 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sat, 7 Jun 2025 22:42:43 +0200 Subject: [PATCH 27/56] chore(spack-repo): Enable HPX_WITH_THREAD_IDLE_RATES by default as recommended by rostam. --- spack-repo/packages/hpx/package.py | 1 + 1 file changed, 1 insertion(+) diff --git a/spack-repo/packages/hpx/package.py b/spack-repo/packages/hpx/package.py index 1f9e5cc3..1449e9b4 100644 --- a/spack-repo/packages/hpx/package.py +++ b/spack-repo/packages/hpx/package.py @@ -257,6 +257,7 @@ def cmake_args(self): self.define_from_variant("HPX_WITH_GENERIC_CONTEXT_COROUTINES", "generic_coroutines"), self.define("BOOST_ROOT", spec["boost"].prefix), self.define("HWLOC_ROOT", spec["hwloc"].prefix), + self.define("HPX_WITH_THREAD_IDLE_RATES", True), self.define("HPX_WITH_BOOST_ALL_DYNAMIC_LINK", True), self.define("BUILD_SHARED_LIBS", True), self.define("HPX_DATASTRUCTURES_WITH_ADAPT_STD_TUPLE", False), From 6a7652c7bfd1cb8266e1acd6aad55b6fe73522b3 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Tue, 1 Jul 2025 12:49:48 +0200 Subject: [PATCH 28/56] chore(spack-repo): Support HPX 1.11.0 --- spack-repo/packages/hpx/package.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/spack-repo/packages/hpx/package.py b/spack-repo/packages/hpx/package.py index 1449e9b4..46c368bb 100644 --- a/spack-repo/packages/hpx/package.py +++ b/spack-repo/packages/hpx/package.py @@ -24,6 +24,7 @@ class Hpx(CMakePackage, CudaPackage, ROCmPackage): version("master", branch="master") version("stable", tag="stable", commit="103a7b8e3719a0db948d1abde29de0ff91e070be") + version("1.11.0", sha256="01ec47228a2253b41e318bb09c83325a75021eb6ef3262400fbda30ac7389279") version("1.10.0", sha256="5720ed7d2460fa0b57bd8cb74fa4f70593fe8675463897678160340526ec3c19") version("1.9.1", sha256="1adae9d408388a723277290ddb33c699aa9ea72defadf3f12d4acc913a0ff22d") version("1.9.0", sha256="2a8dca78172fbb15eae5a5e9facf26ab021c845f9c09e61b1912e6cf9e72915a") @@ -173,7 +174,7 @@ class Hpx(CMakePackage, CudaPackage, ROCmPackage): # Asio 1.34.0 removed io_context::work, used by HPX: # https://github.com/chriskohlhoff/asio/commit/a70f2df321ff40c1809773c2c09986745abf8d20. - conflicts("^asio@1.34:", when="@:1.10") + conflicts("^asio@1.34:") # Certain Asio headers don't compile with nvcc from 1.17.0 onwards with # C++17. Starting with CUDA 11.3 they compile again. From f41fd28f9eeba9ce835b0178ae7156c04e45f90e Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Tue, 27 May 2025 00:42:24 +0200 Subject: [PATCH 29/56] feat(examples): Add work-in-progress distributed version --- CMakeLists.txt | 1 + examples/distributed/CMakeLists.txt | 5 + examples/distributed/src/distributed_blas.cpp | 75 ++++ examples/distributed/src/distributed_blas.hpp | 53 +++ .../distributed/src/distributed_cholesky.hpp | 59 +++ examples/distributed/src/distributed_tile.cpp | 17 + examples/distributed/src/distributed_tile.hpp | 160 ++++++++ examples/distributed/src/main.cpp | 371 ++++++++++++++++++ examples/distributed/src/scheduling.hpp | 26 ++ test/CMakeLists.txt | 3 +- test/src/output_correctness.cpp | 53 +-- test/src/test_data.hpp | 54 +++ 12 files changed, 825 insertions(+), 52 deletions(-) create mode 100644 examples/distributed/CMakeLists.txt create mode 100644 examples/distributed/src/distributed_blas.cpp create mode 100644 examples/distributed/src/distributed_blas.hpp create mode 100644 examples/distributed/src/distributed_cholesky.hpp create mode 100644 examples/distributed/src/distributed_tile.cpp create mode 100644 examples/distributed/src/distributed_tile.hpp create mode 100644 examples/distributed/src/main.cpp create mode 100644 examples/distributed/src/scheduling.hpp create mode 100644 test/src/test_data.hpp diff --git a/CMakeLists.txt b/CMakeLists.txt index f6d7cf40..f7f7d0eb 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -120,6 +120,7 @@ endif() if(GPRAT_ENABLE_EXAMPLES) add_subdirectory(examples/gprat_cpp) + add_subdirectory(examples/distributed) endif() if(GPRAT_ENABLE_TESTS) diff --git a/examples/distributed/CMakeLists.txt b/examples/distributed/CMakeLists.txt new file mode 100644 index 00000000..926f3bfc --- /dev/null +++ b/examples/distributed/CMakeLists.txt @@ -0,0 +1,5 @@ +add_executable(gprat_distributed src/main.cpp src/distributed_blas.cpp src/distributed_tile.cpp) +target_compile_features(gprat_distributed PUBLIC cxx_std_20) + +find_package(Boost REQUIRED) +target_link_libraries(gprat_distributed PUBLIC GPRat::core HPX::hpx Boost::boost) diff --git a/examples/distributed/src/distributed_blas.cpp b/examples/distributed/src/distributed_blas.cpp new file mode 100644 index 00000000..113ba081 --- /dev/null +++ b/examples/distributed/src/distributed_blas.cpp @@ -0,0 +1,75 @@ +#include "distributed_blas.hpp" + +#include "cpu/adapter_cblas_fp64.hpp" +#include + +HPX_REGISTER_ACTION_DECLARATION(potrf_distributed_action); +HPX_REGISTER_ACTION_DECLARATION(trsm_distributed_action); +HPX_REGISTER_ACTION_DECLARATION(syrk_distributed_action); +HPX_REGISTER_ACTION_DECLARATION(gemm_distributed_action); + +tile_handle potrf_distributed(const tile_handle &A, int N) +{ + return hpx::dataflow( + hpx::launch::async, + hpx::unwrapping( + [A, N](tile_data tile) + { + inplace::potrf(tile, N); + return tile_handle(hpx::colocated(A.get_id()), tile); + }), + A.get_data()); +} + +tile_handle +trsm_distributed(const tile_handle &L, const tile_handle &A, int N, int M, BLAS_TRANSPOSE transpose_L, BLAS_SIDE side_L) +{ + return hpx::dataflow( + hpx::launch::async, + hpx::unwrapping( + [L, A, N, M, transpose_L, side_L](const tile_data &Ld, tile_data Ad) + { + inplace::trsm(Ld, Ad, N, M, transpose_L, side_L); + return tile_handle(hpx::colocated(A.get_id()), Ad); + }), + L.get_data(), + A.get_data()); +} + +tile_handle syrk_distributed(const tile_handle &A, const tile_handle &B, int N) +{ + return hpx::dataflow( + hpx::launch::async, + hpx::unwrapping( + [A, B, N](tile_data Ad, const tile_data &Bd) + { + inplace::syrk(Ad, Bd, N); + return tile_handle(hpx::colocated(A.get_id()), Ad); + }), + A.get_data(), + B.get_data()); +} + +tile_handle gemm_distributed( + const tile_handle &A, + const tile_handle &B, + const tile_handle &C, + int N, + int M, + int K, + BLAS_TRANSPOSE transpose_A, + BLAS_TRANSPOSE transpose_B) +{ + return hpx::dataflow( + hpx::launch::async, + hpx::unwrapping( + [A, B, C, N, M, K, transpose_A, transpose_B]( + const tile_data &Ad, const tile_data &Bd, tile_data Cd) + { + inplace::gemm(Ad, Bd, Cd, N, M, K, transpose_A, transpose_B); + return tile_handle(hpx::colocated(C.get_id()), Cd); + }), + A.get_data(), + B.get_data(), + C.get_data()); +} diff --git a/examples/distributed/src/distributed_blas.hpp b/examples/distributed/src/distributed_blas.hpp new file mode 100644 index 00000000..f4702300 --- /dev/null +++ b/examples/distributed/src/distributed_blas.hpp @@ -0,0 +1,53 @@ +#pragma once + +#include "cpu/adapter_cblas_fp64.hpp" +#include "distributed_tile.hpp" +#include "scheduling.hpp" +#include + +tile_handle potrf_distributed(const tile_handle &A, int N); +tile_handle trsm_distributed( + const tile_handle &L, const tile_handle &A, int N, int M, BLAS_TRANSPOSE transpose_L, BLAS_SIDE side_L); +tile_handle syrk_distributed(const tile_handle &A, const tile_handle &B, int N); +tile_handle gemm_distributed( + const tile_handle &A, + const tile_handle &B, + const tile_handle &C, + int N, + int M, + int K, + BLAS_TRANSPOSE transpose_A, + BLAS_TRANSPOSE transpose_B); + +HPX_DEFINE_PLAIN_ACTION(potrf_distributed); +HPX_DEFINE_PLAIN_ACTION(trsm_distributed); +HPX_DEFINE_PLAIN_ACTION(syrk_distributed); +HPX_DEFINE_PLAIN_ACTION(gemm_distributed); + +template <> +struct plain_action_for<&inplace::potrf> +{ + using action_type = potrf_distributed_action; + constexpr static std::string_view name = "POTRF"; +}; + +template <> +struct plain_action_for<&inplace::trsm> +{ + using action_type = trsm_distributed_action; + constexpr static std::string_view name = "TRSM"; +}; + +template <> +struct plain_action_for<&inplace::syrk> +{ + using action_type = syrk_distributed_action; + constexpr static std::string_view name = "SYRK"; +}; + +template <> +struct plain_action_for<&inplace::gemm> +{ + using action_type = gemm_distributed_action; + constexpr static std::string_view name = "GEMM"; +}; diff --git a/examples/distributed/src/distributed_cholesky.hpp b/examples/distributed/src/distributed_cholesky.hpp new file mode 100644 index 00000000..098f97b1 --- /dev/null +++ b/examples/distributed/src/distributed_cholesky.hpp @@ -0,0 +1,59 @@ +#pragma once + +#include "distributed_tile.hpp" +#include "scheduling.hpp" +#include + +struct tiled_cholesky_distribution_policy_paap12 +{ + constexpr std::size_t locality_for_tile(std::size_t row, std::size_t col) const + { + return (row + col) % num_localities; + } + + constexpr std::size_t locality_for_POTRF(std::size_t k) const { return (2 * k) % num_localities; } + + constexpr std::size_t locality_for_SYRK(std::size_t m) const { return (2 * m) % num_localities; } + + constexpr std::size_t locality_for_TRSM(std::size_t k, std::size_t m) const { return (k + m) % num_localities; } + + constexpr std::size_t locality_for_GEMM(std::size_t /*k*/, std::size_t m, std::size_t n) const + { + return (m + n) % num_localities; + } + + std::size_t num_localities; +}; + +template +struct tiled_cholesky_scheduler_distributed +{ + using tiled_matrix_handles = std::vector; + + tiled_cholesky_scheduler_distributed() = default; + + [[nodiscard]] schedule_on_locality for_tile(std::size_t row, std::size_t col) const + { + return localities[policy.locality_for_tile(row, col)]; + } + + [[nodiscard]] schedule_on_locality for_POTRF(std::size_t k) const + { + return localities[policy.locality_for_POTRF(k)]; + } + + [[nodiscard]] schedule_on_locality for_SYRK(std::size_t m) const { return localities[policy.locality_for_SYRK(m)]; } + + [[nodiscard]] schedule_on_locality for_TRSM(std::size_t k, std::size_t m) const + { + return localities[policy.locality_for_TRSM(k, m)]; + } + + [[nodiscard]] schedule_on_locality for_GEMM(std::size_t k, std::size_t m, std::size_t n) const + { + return localities[policy.locality_for_GEMM(k, m, n)]; + } + + std::vector localities = hpx::find_all_localities(); + DistPolicy policy{ localities.size() }; +}; diff --git a/examples/distributed/src/distributed_tile.cpp b/examples/distributed/src/distributed_tile.cpp new file mode 100644 index 00000000..9c716b58 --- /dev/null +++ b/examples/distributed/src/distributed_tile.cpp @@ -0,0 +1,17 @@ +#include "distributed_tile.hpp" + +// The macros below are necessary to generate the code required for exposing +// our partition type remotely. +// +// HPX_REGISTER_COMPONENT() exposes the component creation +// through hpx::new_<>(). +typedef hpx::components::component tile_server_type; +HPX_REGISTER_COMPONENT(tile_server_type, tile_server) + +// HPX_REGISTER_ACTION() exposes the component member function for remote +// invocation. +typedef tile_server::get_data_action get_data_action; +HPX_REGISTER_ACTION(get_data_action) + +typedef tile_server::set_data_action set_data_action; +HPX_REGISTER_ACTION(set_data_action) diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp new file mode 100644 index 00000000..3dd4551d --- /dev/null +++ b/examples/distributed/src/distributed_tile.hpp @@ -0,0 +1,160 @@ +#pragma once + +#include +#include +#include +#include +#include +#include +#include +#include + +template +struct tile_data +{ + private: + typedef hpx::serialization::serialize_buffer buffer_type; + + struct hold_reference + { + explicit hold_reference(const buffer_type &data) : + data_(data) + { } + + void operator()(const double *) const { } // no deletion necessary + + buffer_type data_; + }; + + // In case we want pooling down the road... + static T *allocate(std::size_t n) { return new T[n]; } + + static void deallocate(T *p) noexcept { delete[] p; } + + public: + tile_data() = default; + + // Create a new (uninitialized) partition of the given size. + explicit tile_data(std::size_t size) : + data_(allocate(size), size, buffer_type::take, &tile_data::deallocate) + { } + + // Create a partition which acts as a proxy to a part of the embedded array. + // The proxy is assumed to refer to either the left or the right boundary + // element. + tile_data(const tile_data &base, std::size_t offset, std::size_t size) : + data_(base.data_.data() + offset, + size, + buffer_type::reference, + hold_reference(base.data_)) // keep referenced partition alive + { } + + [[nodiscard]] T *data() noexcept { return data_.data(); } + + [[nodiscard]] const T *data() const noexcept { return data_.data(); } + + [[nodiscard]] std::size_t size() const noexcept { return data_.size(); } + + // ReSharper disable once CppNonExplicitConversionOperator + operator std::span() noexcept { return { data_.data(), data_.size() }; } // NOLINT(*-explicit-constructor) + + // ReSharper disable once CppNonExplicitConversionOperator + operator std::span() const noexcept // NOLINT(*-explicit-constructor) + { + return { data_.data(), data_.size() }; + } + + private: + // Serialization support: even if all of the code below runs on one + // locality only, we need to provide an (empty) implementation for the + // serialization as all arguments passed to actions have to support this. + friend class hpx::serialization::access; + + template + void serialize(Archive &ar, const unsigned int) + { + // clang-format off + ar & data_; + // clang-format on + } + + buffer_type data_; +}; + +/////////////////////////////////////////////////////////////////////////////// +// This is the server side representation of the data. We expose this as a HPX +// component which allows for it to be created and accessed remotely through +// a global address (hpx::id_type). +struct tile_server : hpx::components::component_base +{ + // construct new instances + tile_server() = default; + + explicit tile_server(const tile_data &data) : + data_(data) + { } + + tile_data get_data() const { return data_; } + + void set_data(const tile_data &data) { data_ = data; } + + // Every member function that has to be invoked remotely needs to be + // wrapped into a component action. + HPX_DEFINE_COMPONENT_DIRECT_ACTION(tile_server, get_data, get_data_action) + HPX_DEFINE_COMPONENT_DIRECT_ACTION(tile_server, set_data, set_data_action) + + private: + tile_data data_; +}; + +HPX_REGISTER_ACTION_DECLARATION(tile_server::get_data_action, get_data_action); +HPX_REGISTER_ACTION_DECLARATION(tile_server::set_data_action, set_data_action); + +/////////////////////////////////////////////////////////////////////////////// +// This is a client side helper class allowing to hide some of the tedious +// boilerplate while referencing a remote partition. +struct tile_handle : hpx::components::client_base +{ + typedef hpx::components::client_base base_type; + + tile_handle() = default; + + // Create new component on locality 'where' and initialize the held data + tile_handle(hpx::id_type where, const tile_data &data) : + base_type(hpx::new_(where, data)) + { } + + // Create new component on locality 'where' and initialize the held data + template + requires hpx::traits::is_distribution_policy_v tile_handle(const T &policy, const tile_data &data) : + base_type(hpx::new_(policy, data)) + { } + + // Attach a future representing a (possibly remote) partition. + // ReSharper disable once CppNonExplicitConvertingConstructor + tile_handle(hpx::future &&id) noexcept : + base_type(std::move(id)) + { } + + // Unwrap a future (a tile_handle already is a future to the + // id of the referenced object, thus unwrapping accesses this inner future). + // ReSharper disable once CppNonExplicitConvertingConstructor + tile_handle(hpx::future &&c) noexcept : + base_type(std::move(c)) + { } + + /////////////////////////////////////////////////////////////////////////// + // Invoke the (remote) member function which gives us access to the data. + // This is a pure helper function hiding the async. + [[nodiscard]] hpx::future> get_data() const + { + tile_server::get_data_action act; + return hpx::async(act, get_id()); + } + + [[nodiscard]] hpx::future set_data(const tile_data &data) + { + tile_server::set_data_action act; + return hpx::async(act, get_id(), data); + } +}; diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp new file mode 100644 index 00000000..71dda2d5 --- /dev/null +++ b/examples/distributed/src/main.cpp @@ -0,0 +1,371 @@ +#include "../../test/src/test_data.hpp" +#include "distributed_blas.hpp" +#include "distributed_cholesky.hpp" +#include "distributed_tile.hpp" +#include "cpu/gp_functions.hpp" +#include "gp_kernels.hpp" +#include "gprat_c.hpp" +#include "cpu/tiled_algorithms.hpp" +#include "utils_c.hpp" +#include +#include +#include +#include +#include + +// This is a standalone test, so including this directly is fine. +// Better than having the whole project depend on compiled Boost.Json! +#include + +namespace gprat_hyper +{ + +template +inline void save_construct_data(Archive &ar, const SEKParams *v, const unsigned int) +{ + ar << v->lengthscale; + ar << v->vertical_lengthscale; + ar << v->noise_variance; +} + +template +inline void load_construct_data(Archive &ar, SEKParams *v, const unsigned int) +{ + double lengthscale, vertical_lengthscale, noise_variance; + ar >> lengthscale; + ar >> vertical_lengthscale; + ar >> noise_variance; + + // ::new(ptr) construct new object at given address + hpx::construct_at(v, lengthscale, vertical_lengthscale, noise_variance); +} + +template +void serialize(Archive &ar, SEKParams &pt, const unsigned int) +{ + ar & pt.m_T & pt.w_T; +} + +} // namespace gprat_hyper + +///////////////////////////////////////////////////////// +// Tile generation +double compute_covariance_function(std::size_t n_regressors, + const gprat_hyper::SEKParams &sek_params, + std::span i_input, + std::span j_input) +{ + // k(z_i,z_j) = vertical_lengthscale * exp(-0.5 / lengthscale^2 * (z_i - z_j)^2) + double distance = 0.0; + for (std::size_t k = 0; k < n_regressors; k++) + { + const double z_ik_minus_z_jk = i_input[k] - j_input[k]; + distance += z_ik_minus_z_jk * z_ik_minus_z_jk; + } + + return sek_params.vertical_lengthscale * exp(-0.5 / (sek_params.lengthscale * sek_params.lengthscale) * distance); +} + +tile_data make_covariance_tile( + std::size_t row, + std::size_t col, + std::size_t N, + std::size_t n_regressors, + const gprat_hyper::SEKParams &sek_params, + std::span input) +{ + tile_data tile(N * N); + for (std::size_t i = 0; i < N; i++) + { + std::size_t i_global = N * row + i; + for (std::size_t j = 0; j < N; j++) + { + std::size_t j_global = N * col + j; + + // compute covariance function + auto covariance_function = compute_covariance_function( + n_regressors, sek_params, input.subspan(i_global, n_regressors), input.subspan(j_global, n_regressors)); + if (i_global == j_global) + { + // noise variance on diagonal + covariance_function += sek_params.noise_variance; + } + + tile.data()[i * N + j] = covariance_function; + } + } + return tile; +} + +tile_handle make_covariance_tile_distributed( + std::size_t row, + std::size_t col, + std::size_t N, + std::size_t n_regressors, + const gprat_hyper::SEKParams &sek_params, + std::span input) +{ + return tile_handle(hpx::find_here(), make_covariance_tile(row, col, N, n_regressors, sek_params, input)); +} + +HPX_PLAIN_ACTION(make_covariance_tile_distributed, make_covariance_tile_action) + +template <> +struct plain_action_for<&make_covariance_tile> +{ + using action_type = make_covariance_tile_action; + constexpr static std::string_view name = "gen_tile_covariance"; +}; + +template > +void right_looking_cholesky_tiled( + Scheduler &sched, typename Scheduler::tiled_matrix_handles &ft_tiles, std::size_t N, std::size_t n_tiles) +{ + for (std::size_t k = 0; k < n_tiles; k++) + { + // POTRF: Compute Cholesky factor L + ft_tiles[k * n_tiles + k] = dataflow(sched.for_POTRF(k), ft_tiles[k * n_tiles + k], N); + for (std::size_t m = k + 1; m < n_tiles; m++) + { + // TRSM: Solve X * L^T = A + ft_tiles[m * n_tiles + k] = dataflow( + sched.for_TRSM(k, m), + ft_tiles[k * n_tiles + k], + ft_tiles[m * n_tiles + k], + N, + N, + Blas_trans, + Blas_right); + } + for (std::size_t m = k + 1; m < n_tiles; m++) + { + // SYRK: A = A - B * B^T + ft_tiles[m * n_tiles + m] = + dataflow(sched.for_SYRK(m), ft_tiles[m * n_tiles + m], ft_tiles[m * n_tiles + k], N); + for (std::size_t n = k + 1; n < m; n++) + { + // GEMM: C = C - A * B^T + ft_tiles[m * n_tiles + n] = dataflow( + sched.for_GEMM(k, m, n), + ft_tiles[m * n_tiles + k], + ft_tiles[n * n_tiles + k], + ft_tiles[m * n_tiles + n], + N, + N, + N, + Blas_no_trans, + Blas_trans); + } + } + } +} + +template > +std::vector> +cholesky_hpx(Scheduler &sched, + std::span training_input, + const gprat_hyper::SEKParams &sek_params, + std::size_t n_tiles, + std::size_t n_tile_size, + std::size_t n_regressors) +{ + typename Scheduler::tiled_matrix_handles tiles(n_tiles * n_tiles); // Tiled covariance matrix + + /////////////////////////////////////////////////////////////////////////// + // Launch asynchronous assembly + // std::vector> tile_objs; + // tile_objs.reserve(n_tiles * n_tiles); + + for (std::size_t row = 0; row < n_tiles; row++) + { + for (std::size_t col = 0; col <= row; col++) + { + tiles[row * n_tiles + col] = dataflow( + sched.for_tile(row, col), row, col, n_tile_size, n_regressors, sek_params, training_input); + } + } + + /////////////////////////////////////////////////////////////////////////// + // Launch asynchronous Cholesky decomposition: K = L * L^T + right_looking_cholesky_tiled(sched, tiles, n_tile_size, n_tiles); + + /////////////////////////////////////////////////////////////////////////// + // Synchronize + std::vector> result(n_tiles * n_tiles); + for (std::size_t i = 0; i < n_tiles; i++) + { + for (std::size_t j = 0; j <= i; j++) + { + result[i * n_tiles + j] = tiles[i * n_tiles + j].get_data().get(); + } + } + return result; +} + +gprat_results load_test_data_results(const std::string &filename) +{ + std::ifstream ifs(filename); + if (!ifs.fail()) + { + using iterator_type = std::istreambuf_iterator; + const std::string content(iterator_type{ ifs }, iterator_type{}); + return boost::json::value_to(boost::json::parse(content)); + } + throw std::runtime_error("Failed to load " + filename); +} + +void check_data(const std::vector> &expected, const std::vector> &actual) +{ + if (expected.size() != actual.size()) + { + throw std::runtime_error("expected.size() != actual.size()"); + } + if (expected[0].size() != actual[0].size()) + { + throw std::runtime_error("expected[0].size() != actual[0].size()"); + } + + constexpr double margin = 0.00001; + for (std::size_t i = 0; i < expected.size(); i++) + { + const std::span actual_data = actual[i]; + for (std::size_t j = 0; j < expected[i].size(); j++) + { + const auto &expected_value = expected[i][j]; + const auto &actual_value = actual_data[j]; + + const bool is_in_range = + (expected_value + margin >= actual_value) && (actual_value + margin >= expected_value); + if (!is_in_range) + { + std::cerr << "MISMATCH at " << i << " " << j << " " << expected_value << " !~= " << actual_value; + } + } + } +} + +void run(hpx::program_options::variables_map &vm) +{ + ///////////////////// + /////// configuration + std::size_t START = vm["start"].as(); + std::size_t END = vm["end"].as(); + std::size_t STEP = vm["step"].as(); + std::size_t LOOP = vm["loop"].as(); + const int OPT_ITER = vm["opt_iter"].as(); + + int n_test = 1024; + const std::size_t n_tiles = vm["tiles"].as(); + const std::size_t n_reg = vm["regressors"].as(); + + const auto &train_path = vm["train_x_path"].as(); + const auto &out_path = vm["train_y_path"].as(); + const auto &test_path = vm["test_path"].as(); + //const auto test_results = load_test_data_results(vm["test_results_path"].as()); + + tiled_cholesky_scheduler_distributed scheduler; + + for (std::size_t start = START; start <= END; start = start * STEP) + { + int n_train = static_cast(start); + for (std::size_t l = 0; l < LOOP; l++) + { + auto start_total = std::chrono::high_resolution_clock::now(); + + // Compute tile sizes and number of predict tiles + int tile_size = utils::compute_train_tile_size(n_train, n_tiles); + auto result = utils::compute_test_tiles(n_test, n_tiles, tile_size); + ///////////////////// + ///// hyperparams + gprat_hyper::AdamParams hpar = { 0.1, 0.9, 0.999, 1e-8, OPT_ITER }; + + ///////////////////// + ////// data loading + gprat::GP_data training_input(train_path, n_train, n_reg); + gprat::GP_data training_output(out_path, n_train, n_reg); + gprat::GP_data test_input(test_path, n_test, n_reg); + + ///////////////////// + ///// GP + auto start_init = std::chrono::high_resolution_clock::now(); + std::vector trainable = { true, true, true }; + gprat::GP gp( + training_input.data, training_output.data, n_tiles, tile_size, n_reg, { 1.0, 1.0, 0.1 }, trainable); + auto end_init = std::chrono::high_resolution_clock::now(); + std::chrono::duration init_time = end_init - start_init; + + // Measure the time taken to execute gp.cholesky(); + auto start_cholesky = std::chrono::high_resolution_clock::now(); + + const auto cholesky = + cholesky_hpx(scheduler, training_input.data, { 1.0, 1.0, 0.1 }, n_tiles, tile_size, n_reg); + + auto end_cholesky = std::chrono::high_resolution_clock::now(); + std::chrono::duration cholesky_time = end_cholesky - start_cholesky; + + auto end_total = std::chrono::high_resolution_clock::now(); + std::chrono::duration total_time = end_total - start_total; + + // Save parameters and times to a .txt file with a header + std::ofstream outfile("output-distributed.csv", std::ios::app); // Append mode + if (outfile.tellp() == 0) + { + // If file is empty, write the header + outfile << "Cores,N_train,N_test,N_tiles,N_regressor,Opt_iter,Total_time,Init_time,Cholesky_time," + "Opt_time,Pred_Uncer_time,Pred_Full_time,Pred_time,N_loop\n"; + } + outfile << hpx::get_locality_id() << "," << n_train << "," << n_test << "," << n_tiles << "," << n_reg + << "," << OPT_ITER << "," << total_time.count() << "," << init_time.count() << "," + << cholesky_time.count() << "," << 0 << "," << 0 << "," << 0 << "," << 0 << "," << l << "\n"; + outfile.close(); + + //check_data(test_results.choleksy, cholesky); + } + } + std::cerr << "DONE!" << std::endl; +} + +int hpx_main(hpx::program_options::variables_map &vm) +{ + try + { + run(vm); + } + catch (const std::exception &e) + { + std::cerr << e.what() << std::endl; + } + return hpx::finalize(); +} + +int main(int argc, char *argv[]) +{ + namespace po = hpx::program_options; + po::options_description desc("Allowed options"); +#define BASE_DIR "../../../../" + // clang-format off + desc.add_options() + ("help", "produce help message") + ("train_x_path", po::value()->default_value(BASE_DIR "data/data_1024/training_input.txt"), "training data (x)") + ("train_y_path", po::value()->default_value(BASE_DIR "data/data_1024/training_output.txt"), "training data (y)") + ("test_path", po::value()->default_value(BASE_DIR "data/data_1024/test_input.txt"), "test data") + //("test_results_path", po::value()->default_value(BASE_DIR "data/data_1024/output.json"), "test data results") + ("tiles", po::value()->default_value(16), "tiles per dimension") + ("regressors", po::value()->default_value(8), "num regressors") + ("start", po::value()->default_value(128), "Starting number of training samples") + ("end", po::value()->default_value(128), "End number of training samples") + ("step", po::value()->default_value(2), "Increment of training samples") + ("loop", po::value()->default_value(1), "Number of iterations to be performed for each number of training samples") + ("opt_iter", po::value()->default_value(3), "Number of optimization iterations*/") + ; +// clang-format on +#undef BASE_DIR + + hpx::init_params init_args; + init_args.desc_cmdline = desc; + // If example requires to run hpx_main on all localities + // std::vector const cfg = {"hpx.run_hpx_main!=1"}; + // init_args.cfg = cfg; + // Run HPX main + return hpx::init(argc, argv, init_args); +} diff --git a/examples/distributed/src/scheduling.hpp b/examples/distributed/src/scheduling.hpp new file mode 100644 index 00000000..abf13fb6 --- /dev/null +++ b/examples/distributed/src/scheduling.hpp @@ -0,0 +1,26 @@ +#pragma once + +#include +#include + +template +struct plain_action_for; + +// This is a simple tag-type like construct that exists solely so we automatically pick the right dataflow() overload. +struct schedule_on_locality +{ + // conversion is intended here, we don't want people to actually spell this type out + // ReSharper disable once CppNonExplicitConvertingConstructor + schedule_on_locality(const hpx::id_type &where) : + where(where) + { } + + hpx::id_type where; +}; + +template +decltype(auto) dataflow(const schedule_on_locality &on, Args &&...args) +{ + typename plain_action_for::action_type act; + return hpx::dataflow(act, on.where, args...); +} diff --git a/test/CMakeLists.txt b/test/CMakeLists.txt index 9618627f..65ca56c7 100644 --- a/test/CMakeLists.txt +++ b/test/CMakeLists.txt @@ -25,7 +25,8 @@ find_package(Boost REQUIRED) # ---- Tests ---- -add_executable(GPRat_test_output_correctness src/output_correctness.cpp) +add_executable(GPRat_test_output_correctness src/test_data.hpp + src/output_correctness.cpp) target_link_libraries(GPRat_test_output_correctness PRIVATE GPRat::core Catch2::Catch2WithMain Boost::boost) target_compile_features(GPRat_test_output_correctness PRIVATE cxx_std_17) diff --git a/test/src/output_correctness.cpp b/test/src/output_correctness.cpp index 1e7ca8fc..f924e8f8 100644 --- a/test/src/output_correctness.cpp +++ b/test/src/output_correctness.cpp @@ -1,3 +1,4 @@ +#include "test_data.hpp" #include "gprat/gprat.hpp" #include "gprat/utils.hpp" @@ -13,34 +14,6 @@ #include #include -// Struct containing all results we'd like to compare -struct gprat_results -{ - std::vector> choleksy; - std::vector losses; - std::vector> sum; - std::vector> full; - std::vector pred; - std::vector> sum_no_optimize; - std::vector> full_no_optimize; - std::vector pred_no_optimize; -}; - -// The following two functions are for JSON (de-)serialization -void tag_invoke(boost::json::value_from_tag, boost::json::value &jv, const gprat_results &results) -{ - jv = { - { "choleksy", boost::json::value_from(results.choleksy) }, - { "losses", boost::json::value_from(results.losses) }, - { "sum", boost::json::value_from(results.sum) }, - { "full", boost::json::value_from(results.full) }, - { "pred", boost::json::value_from(results.pred) }, - { "sum_no_optimize", boost::json::value_from(results.sum_no_optimize) }, - { "full_no_optimize", boost::json::value_from(results.full_no_optimize) }, - { "pred_no_optimize", boost::json::value_from(results.pred_no_optimize) }, - }; -} - template std::vector to_vector(const gprat::const_tile_data &data) { @@ -71,28 +44,6 @@ std::vector> to_vector(const std::vector -inline void extract(const boost::json::object &obj, T &t, std::string_view key) -{ - t = boost::json::value_to(obj.at(key)); -} - -gprat_results tag_invoke(boost::json::value_to_tag, const boost::json::value &jv) -{ - gprat_results results; - const auto &obj = jv.as_object(); - extract(obj, results.choleksy, "choleksy"); - extract(obj, results.losses, "losses"); - extract(obj, results.sum, "sum"); - extract(obj, results.full, "full"); - extract(obj, results.pred, "pred"); - extract(obj, results.sum_no_optimize, "sum_no_optimize"); - extract(obj, results.full_no_optimize, "full_no_optimize"); - extract(obj, results.pred_no_optimize, "pred_no_optimize"); - return results; -} - // This logic is basically equivalent to the GPRat C++ example (for now). gprat_results run_on_data_cpu(const std::string &train_path, const std::string &out_path, const std::string &test_path) { @@ -284,7 +235,7 @@ TEST_CASE("GP GPU results match known-good values (no loss)", "[integration][gpu { if (!gprat::compiled_with_cuda()) { - WARN("CUDA not available — skipping GPU test."); + WARN("CUDA not available — skipping GPU test."); return; } diff --git a/test/src/test_data.hpp b/test/src/test_data.hpp new file mode 100644 index 00000000..aa759446 --- /dev/null +++ b/test/src/test_data.hpp @@ -0,0 +1,54 @@ +#pragma once + +#include +#include + +// Struct containing all results we'd like to compare +struct gprat_results +{ + std::vector> choleksy; + std::vector losses; + std::vector> sum; + std::vector> full; + std::vector pred; + std::vector> sum_no_optimize; + std::vector> full_no_optimize; + std::vector pred_no_optimize; +}; + +// The following two functions are for JSON (de-)serialization +inline void tag_invoke(boost::json::value_from_tag, boost::json::value &jv, const gprat_results &results) +{ + jv = { + { "choleksy", boost::json::value_from(results.choleksy) }, + { "losses", boost::json::value_from(results.losses) }, + { "sum", boost::json::value_from(results.sum) }, + { "full", boost::json::value_from(results.full) }, + { "pred", boost::json::value_from(results.pred) }, + { "sum_no_optimize", boost::json::value_from(results.sum_no_optimize) }, + { "full_no_optimize", boost::json::value_from(results.full_no_optimize) }, + { "pred_no_optimize", boost::json::value_from(results.pred_no_optimize) }, + }; +} + +// This helper function deduces the type and assigns the value with the matching key +template +BOOST_FORCEINLINE void extract(const boost::json::object &obj, T &t, std::string_view key) +{ + t = boost::json::value_to(obj.at(key)); +} + +inline gprat_results tag_invoke(boost::json::value_to_tag, const boost::json::value &jv) +{ + gprat_results results; + const auto &obj = jv.as_object(); + extract(obj, results.choleksy, "choleksy"); + extract(obj, results.losses, "losses"); + extract(obj, results.sum, "sum"); + extract(obj, results.full, "full"); + extract(obj, results.pred, "pred"); + extract(obj, results.sum_no_optimize, "sum_no_optimize"); + extract(obj, results.full_no_optimize, "full_no_optimize"); + extract(obj, results.pred_no_optimize, "pred_no_optimize"); + return results; +} From 44278892699d3294de792494a72f35c6d7b469d2 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Wed, 28 May 2025 04:56:29 +0200 Subject: [PATCH 30/56] feat(examples): Implement local data caching for immutable tile components We can easily cache that in the component's client stub without having to touch the API for now. --- examples/distributed/CMakeLists.txt | 9 ++- examples/distributed/src/distributed_tile.cpp | 20 ++++++ examples/distributed/src/distributed_tile.hpp | 36 ++++++++--- examples/distributed/src/main.cpp | 62 ++++++++++++++----- 4 files changed, 100 insertions(+), 27 deletions(-) diff --git a/examples/distributed/CMakeLists.txt b/examples/distributed/CMakeLists.txt index 926f3bfc..3c0275b2 100644 --- a/examples/distributed/CMakeLists.txt +++ b/examples/distributed/CMakeLists.txt @@ -1,5 +1,10 @@ -add_executable(gprat_distributed src/main.cpp src/distributed_blas.cpp src/distributed_tile.cpp) +add_executable(gprat_distributed src/main.cpp src/distributed_blas.cpp + src/distributed_tile.cpp) target_compile_features(gprat_distributed PUBLIC cxx_std_20) find_package(Boost REQUIRED) -target_link_libraries(gprat_distributed PUBLIC GPRat::core HPX::hpx Boost::boost) +target_link_libraries(gprat_distributed PUBLIC GPRat::core HPX::hpx + Boost::boost) + +set_target_properties(gprat_distributed PROPERTIES VS_DEBUGGER_WORKING_DIRECTORY + "${CMAKE_SOURCE_DIR}") diff --git a/examples/distributed/src/distributed_tile.cpp b/examples/distributed/src/distributed_tile.cpp index 9c716b58..a8968c83 100644 --- a/examples/distributed/src/distributed_tile.cpp +++ b/examples/distributed/src/distributed_tile.cpp @@ -1,5 +1,25 @@ #include "distributed_tile.hpp" +namespace hpx::util +{ + +// This is explicitly instantiated to ensure that the id is stable +// across shared libraries. +template <> +extra_data_id_type extra_data_helper::id() noexcept +{ + static std::uint8_t id = 0; + return &id; +} + +template <> +void extra_data_helper::reset(tile_handle_cache *data) noexcept +{ + data->cached_data.reset(); +} + +} // namespace hpx::util + // The macros below are necessary to generate the code required for exposing // our partition type remotely. // diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index 3dd4551d..ba4c21f6 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -1,9 +1,9 @@ #pragma once -#include #include -#include +#include #include +#include #include #include #include @@ -113,14 +113,19 @@ HPX_REGISTER_ACTION_DECLARATION(tile_server::set_data_action, set_data_action); /////////////////////////////////////////////////////////////////////////////// // This is a client side helper class allowing to hide some of the tedious // boilerplate while referencing a remote partition. -struct tile_handle : hpx::components::client_base +struct tile_handle_cache { - typedef hpx::components::client_base base_type; + hpx::optional> cached_data; +}; + +struct tile_handle : hpx::components::client_base +{ + using base_type = hpx::components::client_base; tile_handle() = default; // Create new component on locality 'where' and initialize the held data - tile_handle(hpx::id_type where, const tile_data &data) : + tile_handle(const hpx::id_type &where, const tile_data &data) : base_type(hpx::new_(where, data)) { } @@ -144,12 +149,27 @@ struct tile_handle : hpx::components::client_base { } /////////////////////////////////////////////////////////////////////////// - // Invoke the (remote) member function which gives us access to the data. - // This is a pure helper function hiding the async. + // tile's are immutable for now, meaning we can cache them on first use. + // TODO: profile whether this strategy (new components on mutation) beats mutable tile state. [[nodiscard]] hpx::future> get_data() const { + if (const auto data_ptr = try_get_extra_data()) + { + if (data_ptr->cached_data) + { + return hpx::make_ready_future(*data_ptr->cached_data); + } + } + tile_server::get_data_action act; - return hpx::async(act, get_id()); + return hpx::dataflow( + [self = const_cast(*this)](hpx::future> f) mutable + { + auto data = f.get(); + self.get_extra_data().cached_data = data; + return std::move(data); + }, + hpx::async(act, get_id())); } [[nodiscard]] hpx::future set_data(const tile_data &data) diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index 71dda2d5..a86e1c29 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -180,8 +180,9 @@ cholesky_hpx(Scheduler &sched, { for (std::size_t col = 0; col <= row; col++) { - tiles[row * n_tiles + col] = dataflow( - sched.for_tile(row, col), row, col, n_tile_size, n_regressors, sek_params, training_input); + tiles[row * n_tiles + col] = + tile_handle(sched.for_tile(row, col).where, + make_covariance_tile(row, col, n_tile_size, n_regressors, sek_params, training_input)); } } @@ -214,34 +215,45 @@ gprat_results load_test_data_results(const std::string &filename) throw std::runtime_error("Failed to load " + filename); } -void check_data(const std::vector> &expected, const std::vector> &actual) +void validate_two_dim_result(const std::vector> &expected, + const std::vector> &actual) { if (expected.size() != actual.size()) { throw std::runtime_error("expected.size() != actual.size()"); } - if (expected[0].size() != actual[0].size()) - { - throw std::runtime_error("expected[0].size() != actual[0].size()"); - } constexpr double margin = 0.00001; + bool is_valid = true; for (std::size_t i = 0; i < expected.size(); i++) { + if (expected[i].size() != actual[i].size()) + { + throw std::runtime_error("expected[i].size() != actual[i].size(): i = " + std::to_string(i)); + } + const std::span actual_data = actual[i]; for (std::size_t j = 0; j < expected[i].size(); j++) { const auto &expected_value = expected[i][j]; const auto &actual_value = actual_data[j]; + // XXX: no std::abs(expected - actual) due to infinity const bool is_in_range = (expected_value + margin >= actual_value) && (actual_value + margin >= expected_value); if (!is_in_range) { - std::cerr << "MISMATCH at " << i << " " << j << " " << expected_value << " !~= " << actual_value; + std::cerr << "MISMATCH at " << i << " " << j << " " << expected_value << " !~= " << actual_value + << std::endl; + is_valid = false; } } } + + if (!is_valid) + { + throw std::runtime_error("Invalid results (see stderr for details)"); + } } void run(hpx::program_options::variables_map &vm) @@ -261,7 +273,15 @@ void run(hpx::program_options::variables_map &vm) const auto &train_path = vm["train_x_path"].as(); const auto &out_path = vm["train_y_path"].as(); const auto &test_path = vm["test_path"].as(); - //const auto test_results = load_test_data_results(vm["test_results_path"].as()); + + std::optional test_results; + // XXX: cannot use contains() because it's not exported by HPX program_options + // ReSharper disable once CppUseAssociativeContains + if (vm.find("test_results_path") != vm.end()) + { + test_results = load_test_data_results(vm["test_results_path"].as()); + std::cerr << "We have comparison data!" << std::endl; + } tiled_cholesky_scheduler_distributed scheduler; @@ -319,7 +339,10 @@ void run(hpx::program_options::variables_map &vm) << cholesky_time.count() << "," << 0 << "," << 0 << "," << 0 << "," << 0 << "," << l << "\n"; outfile.close(); - //check_data(test_results.choleksy, cholesky); + if (test_results) + { + validate_two_dim_result(test_results->choleksy, cholesky); + } } } std::cerr << "DONE!" << std::endl; @@ -327,6 +350,12 @@ void run(hpx::program_options::variables_map &vm) int hpx_main(hpx::program_options::variables_map &vm) { + std::cerr << "OS Threads: " << hpx::get_os_thread_count() << std::endl; + std::cerr << "All localities: " << hpx::get_num_localities().get() << std::endl; + std::cerr << "Root locality: " << hpx::find_root_locality() << std::endl; + std::cerr << "This locality: " << hpx::find_here() << std::endl; + std::cerr << "Remote localities: " << hpx::find_remote_localities().size() << std::endl; + try { run(vm); @@ -342,14 +371,14 @@ int main(int argc, char *argv[]) { namespace po = hpx::program_options; po::options_description desc("Allowed options"); -#define BASE_DIR "../../../../" + // clang-format off desc.add_options() ("help", "produce help message") - ("train_x_path", po::value()->default_value(BASE_DIR "data/data_1024/training_input.txt"), "training data (x)") - ("train_y_path", po::value()->default_value(BASE_DIR "data/data_1024/training_output.txt"), "training data (y)") - ("test_path", po::value()->default_value(BASE_DIR "data/data_1024/test_input.txt"), "test data") - //("test_results_path", po::value()->default_value(BASE_DIR "data/data_1024/output.json"), "test data results") + ("train_x_path", po::value()->default_value("data/data_1024/training_input.txt"), "training data (x)") + ("train_y_path", po::value()->default_value("data/data_1024/training_output.txt"), "training data (y)") + ("test_path", po::value()->default_value("data/data_1024/test_input.txt"), "test data") + ("test_results_path", po::value(), "test data results") ("tiles", po::value()->default_value(16), "tiles per dimension") ("regressors", po::value()->default_value(8), "num regressors") ("start", po::value()->default_value(128), "Starting number of training samples") @@ -358,8 +387,7 @@ int main(int argc, char *argv[]) ("loop", po::value()->default_value(1), "Number of iterations to be performed for each number of training samples") ("opt_iter", po::value()->default_value(3), "Number of optimization iterations*/") ; -// clang-format on -#undef BASE_DIR + // clang-format on hpx::init_params init_args; init_args.desc_cmdline = desc; From 5647e89ccd1d113ba1d16a31104d92caf14d6832 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Wed, 28 May 2025 05:21:21 +0200 Subject: [PATCH 31/56] chore(examples): Make timing csv filename configurable --- examples/distributed/src/main.cpp | 5 +++-- examples/gprat_cpp/src/execute.cpp | 7 +++++-- 2 files changed, 8 insertions(+), 4 deletions(-) diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index a86e1c29..150938eb 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -327,7 +327,7 @@ void run(hpx::program_options::variables_map &vm) std::chrono::duration total_time = end_total - start_total; // Save parameters and times to a .txt file with a header - std::ofstream outfile("output-distributed.csv", std::ios::app); // Append mode + std::ofstream outfile(vm["timings_csv"].as(), std::ios::app); if (outfile.tellp() == 0) { // If file is empty, write the header @@ -378,7 +378,8 @@ int main(int argc, char *argv[]) ("train_x_path", po::value()->default_value("data/data_1024/training_input.txt"), "training data (x)") ("train_y_path", po::value()->default_value("data/data_1024/training_output.txt"), "training data (y)") ("test_path", po::value()->default_value("data/data_1024/test_input.txt"), "test data") - ("test_results_path", po::value(), "test data results") + ("test_results_path", po::value(), "test data results to validate results with") + ("timings_csv", po::value()->default_value("timings.csv"), "output timing reports") ("tiles", po::value()->default_value(16), "tiles per dimension") ("regressors", po::value()->default_value(8), "num regressors") ("start", po::value()->default_value(128), "Starting number of training samples") diff --git a/examples/gprat_cpp/src/execute.cpp b/examples/gprat_cpp/src/execute.cpp index 7089155e..a2eae0b3 100644 --- a/examples/gprat_cpp/src/execute.cpp +++ b/examples/gprat_cpp/src/execute.cpp @@ -15,6 +15,7 @@ int main(int argc, char *argv[]) ("train_x_path", po::value()->default_value("../../../data/data_1024/training_input.txt"), "training data (x)") ("train_y_path", po::value()->default_value("../../../data/data_1024/training_output.txt"), "training data (y)") ("test_path", po::value()->default_value("../../../data/data_1024/test_input.txt"), "test data") + ("timings_csv", po::value()->default_value("output.csv"), "output timing data") ("tiles", po::value()->default_value(16), "tiles per dimension") ("regressors", po::value()->default_value(8), "num regressors") ("start-cores", po::value()->default_value(2), "num CPUs to start with") @@ -31,7 +32,9 @@ int main(int argc, char *argv[]) po::store(po::parse_command_line(argc, argv, desc), vm); po::notify(vm); - if (vm.contains("help")) + // XXX: cannot use contains() because it's not exported by HPX program_options + // ReSharper disable once CppUseAssociativeContains + if (vm.find("help") != vm.end()) { std::cout << desc << "\n"; return 1; @@ -205,7 +208,7 @@ int main(int argc, char *argv[]) auto total_time = end_total - start_total; // Save parameters and times to a .txt file with a header - std::ofstream outfile("output.csv", std::ios::app); // Append mode + std::ofstream outfile(vm["timings_csv"].as(), std::ios::app); // Append mode if (outfile.tellp() == 0) { // If file is empty, write the header From 266e016fe4b0b2b2717056aa6eb35d6792e928bd Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Wed, 4 Jun 2025 00:25:11 +0200 Subject: [PATCH 32/56] refactor!(examples): Try another local caching mechanism It seems the extra data approach doesn't work. --- examples/distributed/src/distributed_tile.cpp | 15 ------ examples/distributed/src/distributed_tile.hpp | 49 +++++++++++++++---- 2 files changed, 40 insertions(+), 24 deletions(-) diff --git a/examples/distributed/src/distributed_tile.cpp b/examples/distributed/src/distributed_tile.cpp index a8968c83..8fac5903 100644 --- a/examples/distributed/src/distributed_tile.cpp +++ b/examples/distributed/src/distributed_tile.cpp @@ -3,21 +3,6 @@ namespace hpx::util { -// This is explicitly instantiated to ensure that the id is stable -// across shared libraries. -template <> -extra_data_id_type extra_data_helper::id() noexcept -{ - static std::uint8_t id = 0; - return &id; -} - -template <> -void extra_data_helper::reset(tile_handle_cache *data) noexcept -{ - data->cached_data.reset(); -} - } // namespace hpx::util // The macros below are necessary to generate the code required for exposing diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index ba4c21f6..92a4315e 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -2,6 +2,7 @@ #include #include +#include #include #include #include @@ -113,14 +114,41 @@ HPX_REGISTER_ACTION_DECLARATION(tile_server::set_data_action, set_data_action); /////////////////////////////////////////////////////////////////////////////// // This is a client side helper class allowing to hide some of the tedious // boilerplate while referencing a remote partition. -struct tile_handle_cache + +class tile_cache { - hpx::optional> cached_data; + public: + tile_cache() : + cache_(16) + { } + + bool try_get(const hpx::naming::gid_type &key, tile_data &cached_data) + { + std::lock_guard g(mutex_); + hpx::naming::gid_type unused; + return cache_.get_entry(key, unused, cached_data); + } + + void insert(const hpx::naming::gid_type &key, const tile_data &data) + { + std::lock_guard g(mutex_); + cache_.insert(key, data); + } + + private: + hpx::mutex mutex_; + hpx::util::cache::lru_cache> cache_; }; -struct tile_handle : hpx::components::client_base +inline tile_cache &get_tile_cache() +{ + static tile_cache cache; + return cache; +} + +struct tile_handle : hpx::components::client_base { - using base_type = hpx::components::client_base; + using base_type = hpx::components::client_base; tile_handle() = default; @@ -153,20 +181,23 @@ struct tile_handle : hpx::components::client_base> get_data() const { - if (const auto data_ptr = try_get_extra_data()) + hpx::naming::gid_type current_gid = get().get_gid(); { - if (data_ptr->cached_data) + auto &cache = get_tile_cache(); + tile_data cached_data; + if (cache.try_get(current_gid, cached_data)) { - return hpx::make_ready_future(*data_ptr->cached_data); + return hpx::make_ready_future(cached_data); } } tile_server::get_data_action act; return hpx::dataflow( - [self = const_cast(*this)](hpx::future> f) mutable + [self = *this, current_gid](hpx::future> f) mutable { + (void) self; // need this ref-counted object to stay alive! auto data = f.get(); - self.get_extra_data().cached_data = data; + get_tile_cache().insert(current_gid, data); return std::move(data); }, hpx::async(act, get_id())); From 3b0788a2fe59bcbe8843021f42024cb504f4c35c Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sun, 15 Jun 2025 07:15:00 +0200 Subject: [PATCH 33/56] feat(examples): Add HPX performance counters for tile cache --- examples/distributed/src/distributed_tile.cpp | 85 ++++++++++++++++++- examples/distributed/src/distributed_tile.hpp | 31 ++++--- examples/distributed/src/main.cpp | 2 + 3 files changed, 100 insertions(+), 18 deletions(-) diff --git a/examples/distributed/src/distributed_tile.cpp b/examples/distributed/src/distributed_tile.cpp index 8fac5903..50844ca3 100644 --- a/examples/distributed/src/distributed_tile.cpp +++ b/examples/distributed/src/distributed_tile.cpp @@ -1,9 +1,90 @@ #include "distributed_tile.hpp" -namespace hpx::util +#include +#include + +tile_cache::tile_cache() : + cache_(16) +{ } + +bool tile_cache::try_get(const hpx::naming::gid_type &key, tile_data &cached_data) +{ + std::lock_guard g(mutex_); + hpx::naming::gid_type unused; + return cache_.get_entry(key, unused, cached_data); +} + +void tile_cache::insert(const hpx::naming::gid_type &key, const tile_data &data) +{ + std::lock_guard g(mutex_); + cache_.insert(key, data); +} + +struct tile_cache_counters +{ + // XXX: you can do this with templates, but it's quite a bit more complicated +#define GPRAT_MAKE_STATISTICS_ACCESSOR(name) \ + static std::uint64_t get_cache_##name(bool reset) \ + { \ + auto &cache = get_tile_cache(); \ + std::lock_guard lock(cache.mutex_); \ + return cache.cache_.get_statistics().name(reset); \ + } + + GPRAT_MAKE_STATISTICS_ACCESSOR(hits); + GPRAT_MAKE_STATISTICS_ACCESSOR(misses); + GPRAT_MAKE_STATISTICS_ACCESSOR(evictions); + GPRAT_MAKE_STATISTICS_ACCESSOR(insertions); + +#undef GPRAT_MAKE_STATISTICS_ACCESSOR +}; + +std::atomic tile_transmission_time(0); + +void record_transmission_time(std::int64_t elapsed_ns) { + HPX_ASSERT(elapsed_ns >= 0); + tile_transmission_time += elapsed_ns; +} -} // namespace hpx::util +std::uint64_t get_transmission_time(bool reset) +{ + return hpx::util::get_and_reset_value(tile_transmission_time, reset); +} + +void register_distributed_tile_counters() +{ + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/hits", + &tile_cache_counters::get_cache_hits, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/misses", + &tile_cache_counters::get_cache_misses, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/evictions", + &tile_cache_counters::get_cache_evictions, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/insertions", + &tile_cache_counters::get_cache_insertions, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/transmission_time", + &get_transmission_time, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); +} // The macros below are necessary to generate the code required for exposing // our partition type remotely. diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index 92a4315e..dba3f086 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -1,5 +1,6 @@ #pragma once +#include #include #include #include @@ -117,27 +118,19 @@ HPX_REGISTER_ACTION_DECLARATION(tile_server::set_data_action, set_data_action); class tile_cache { + friend struct tile_cache_counters; public: - tile_cache() : - cache_(16) - { } + tile_cache(); - bool try_get(const hpx::naming::gid_type &key, tile_data &cached_data) - { - std::lock_guard g(mutex_); - hpx::naming::gid_type unused; - return cache_.get_entry(key, unused, cached_data); - } + bool try_get(const hpx::naming::gid_type &key, tile_data &cached_data); - void insert(const hpx::naming::gid_type &key, const tile_data &data) - { - std::lock_guard g(mutex_); - cache_.insert(key, data); - } + void insert(const hpx::naming::gid_type &key, const tile_data &data); private: hpx::mutex mutex_; - hpx::util::cache::lru_cache> cache_; + hpx::util::cache:: + lru_cache, hpx::util::cache::statistics::local_full_statistics> + cache_; }; inline tile_cache &get_tile_cache() @@ -146,6 +139,9 @@ inline tile_cache &get_tile_cache() return cache; } +void register_distributed_tile_counters(); +void record_transmission_time(std::int64_t elapsed_ns); + struct tile_handle : hpx::components::client_base { using base_type = hpx::components::client_base; @@ -191,12 +187,15 @@ struct tile_handle : hpx::components::client_base } } + const auto start = hpx::chrono::high_resolution_clock::now(); + tile_server::get_data_action act; return hpx::dataflow( - [self = *this, current_gid](hpx::future> f) mutable + [self = *this, current_gid, timer = hpx::chrono::high_resolution_timer()](hpx::future> f) mutable { (void) self; // need this ref-counted object to stay alive! auto data = f.get(); + record_transmission_time(timer.elapsed_nanoseconds()); get_tile_cache().insert(current_gid, data); return std::move(data); }, diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index 150938eb..2ce0f66e 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -369,6 +369,8 @@ int hpx_main(hpx::program_options::variables_map &vm) int main(int argc, char *argv[]) { + hpx::register_startup_function(®ister_distributed_tile_counters); + namespace po = hpx::program_options; po::options_description desc("Allowed options"); From 0dd14b90549dc77cc051a1c2e82db2536c1c82d9 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sun, 15 Jun 2025 22:49:53 +0200 Subject: [PATCH 34/56] feat(examples): Add HPX perf counters for distributed BLAS operations --- examples/distributed/src/distributed_blas.cpp | 34 +++++++++++++++++++ examples/distributed/src/distributed_blas.hpp | 31 +++-------------- examples/distributed/src/main.cpp | 20 +++++------ examples/distributed/src/scheduling.hpp | 32 +++++++++++++++-- 4 files changed, 77 insertions(+), 40 deletions(-) diff --git a/examples/distributed/src/distributed_blas.cpp b/examples/distributed/src/distributed_blas.cpp index 113ba081..9a2f0c4f 100644 --- a/examples/distributed/src/distributed_blas.cpp +++ b/examples/distributed/src/distributed_blas.cpp @@ -2,12 +2,18 @@ #include "cpu/adapter_cblas_fp64.hpp" #include +#include HPX_REGISTER_ACTION_DECLARATION(potrf_distributed_action); HPX_REGISTER_ACTION_DECLARATION(trsm_distributed_action); HPX_REGISTER_ACTION_DECLARATION(syrk_distributed_action); HPX_REGISTER_ACTION_DECLARATION(gemm_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&inplace::potrf); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&inplace::trsm); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&inplace::syrk); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&inplace::gemm); + tile_handle potrf_distributed(const tile_handle &A, int N) { return hpx::dataflow( @@ -73,3 +79,31 @@ tile_handle gemm_distributed( B.get_data(), C.get_data()); } + +void register_distributed_blas_counters() +{ + hpx::performance_counters::install_counter_type( + "/gprat/potrf/time", + get_and_reset_plain_action_elapsed<&inplace::potrf>, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/trsm/time", + get_and_reset_plain_action_elapsed<&inplace::trsm>, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/syrk/time", + get_and_reset_plain_action_elapsed<&inplace::syrk>, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/gemm/time", + get_and_reset_plain_action_elapsed<&inplace::gemm>, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); +} diff --git a/examples/distributed/src/distributed_blas.hpp b/examples/distributed/src/distributed_blas.hpp index f4702300..867aeac7 100644 --- a/examples/distributed/src/distributed_blas.hpp +++ b/examples/distributed/src/distributed_blas.hpp @@ -24,30 +24,9 @@ HPX_DEFINE_PLAIN_ACTION(trsm_distributed); HPX_DEFINE_PLAIN_ACTION(syrk_distributed); HPX_DEFINE_PLAIN_ACTION(gemm_distributed); -template <> -struct plain_action_for<&inplace::potrf> -{ - using action_type = potrf_distributed_action; - constexpr static std::string_view name = "POTRF"; -}; +GPRAT_DECLARE_PLAIN_ACTION_FOR(&inplace::potrf, potrf_distributed_action, "POTRF"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&inplace::trsm, trsm_distributed_action, "TRSM"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&inplace::syrk, syrk_distributed_action, "SYRK"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&inplace::gemm, gemm_distributed_action, "GEMM"); -template <> -struct plain_action_for<&inplace::trsm> -{ - using action_type = trsm_distributed_action; - constexpr static std::string_view name = "TRSM"; -}; - -template <> -struct plain_action_for<&inplace::syrk> -{ - using action_type = syrk_distributed_action; - constexpr static std::string_view name = "SYRK"; -}; - -template <> -struct plain_action_for<&inplace::gemm> -{ - using action_type = gemm_distributed_action; - constexpr static std::string_view name = "GEMM"; -}; +void register_distributed_blas_counters(); diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index 2ce0f66e..bbf0994a 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -290,7 +290,7 @@ void run(hpx::program_options::variables_map &vm) int n_train = static_cast(start); for (std::size_t l = 0; l < LOOP; l++) { - auto start_total = std::chrono::high_resolution_clock::now(); + hpx::chrono::high_resolution_timer total_timer; // Compute tile sizes and number of predict tiles int tile_size = utils::compute_train_tile_size(n_train, n_tiles); @@ -307,24 +307,19 @@ void run(hpx::program_options::variables_map &vm) ///////////////////// ///// GP - auto start_init = std::chrono::high_resolution_clock::now(); + hpx::chrono::high_resolution_timer init_timer; std::vector trainable = { true, true, true }; gprat::GP gp( training_input.data, training_output.data, n_tiles, tile_size, n_reg, { 1.0, 1.0, 0.1 }, trainable); - auto end_init = std::chrono::high_resolution_clock::now(); - std::chrono::duration init_time = end_init - start_init; + const auto init_time = init_timer.elapsed(); // Measure the time taken to execute gp.cholesky(); auto start_cholesky = std::chrono::high_resolution_clock::now(); + hpx::chrono::high_resolution_timer cholesky_timer; const auto cholesky = cholesky_hpx(scheduler, training_input.data, { 1.0, 1.0, 0.1 }, n_tiles, tile_size, n_reg); - - auto end_cholesky = std::chrono::high_resolution_clock::now(); - std::chrono::duration cholesky_time = end_cholesky - start_cholesky; - - auto end_total = std::chrono::high_resolution_clock::now(); - std::chrono::duration total_time = end_total - start_total; + const auto cholesky_time = cholesky_timer.elapsed(); // Save parameters and times to a .txt file with a header std::ofstream outfile(vm["timings_csv"].as(), std::ios::app); @@ -335,8 +330,8 @@ void run(hpx::program_options::variables_map &vm) "Opt_time,Pred_Uncer_time,Pred_Full_time,Pred_time,N_loop\n"; } outfile << hpx::get_locality_id() << "," << n_train << "," << n_test << "," << n_tiles << "," << n_reg - << "," << OPT_ITER << "," << total_time.count() << "," << init_time.count() << "," - << cholesky_time.count() << "," << 0 << "," << 0 << "," << 0 << "," << 0 << "," << l << "\n"; + << "," << OPT_ITER << "," << total_timer.elapsed() << "," << init_time << "," << cholesky_time + << "," << 0 << "," << 0 << "," << 0 << "," << 0 << "," << l << "\n"; outfile.close(); if (test_results) @@ -370,6 +365,7 @@ int hpx_main(hpx::program_options::variables_map &vm) int main(int argc, char *argv[]) { hpx::register_startup_function(®ister_distributed_tile_counters); + hpx::register_startup_function(®ister_distributed_blas_counters); namespace po = hpx::program_options; po::options_description desc("Allowed options"); diff --git a/examples/distributed/src/scheduling.hpp b/examples/distributed/src/scheduling.hpp index abf13fb6..12dcef88 100644 --- a/examples/distributed/src/scheduling.hpp +++ b/examples/distributed/src/scheduling.hpp @@ -1,12 +1,32 @@ #pragma once #include +#include #include +#include template struct plain_action_for; -// This is a simple tag-type like construct that exists solely so we automatically pick the right dataflow() overload. +#define GPRAT_DECLARE_PLAIN_ACTION_FOR(local_function, action, friendly_name) \ + template <> \ + struct plain_action_for \ + { \ + using action_type = action; \ + constexpr static std::string_view name = friendly_name; \ + static std::atomic elapsed_ns_in_action; \ + } + +#define GPRAT_DEFINE_PLAIN_ACTION_FOR(local_function) \ + std::atomic plain_action_for::elapsed_ns_in_action(0) + +template +std::uint64_t get_and_reset_plain_action_elapsed(bool reset) +{ + return hpx::util::get_and_reset_value(plain_action_for::elapsed_ns_in_action, reset); +} + +// This is a simple tag-type like construct that exists solely, so we automatically pick the right dataflow() overload. struct schedule_on_locality { // conversion is intended here, we don't want people to actually spell this type out @@ -22,5 +42,13 @@ template decltype(auto) dataflow(const schedule_on_locality &on, Args &&...args) { typename plain_action_for::action_type act; - return hpx::dataflow(act, on.where, args...); + return hpx::dataflow( + [timer = hpx::chrono::high_resolution_timer()](auto &&r) + { + const auto elapsed = timer.elapsed_nanoseconds(); + HPX_ASSERT(elapsed >= 0); + plain_action_for::elapsed_ns_in_action += elapsed; + return r.get(); + }, + hpx::dataflow(act, on.where, args...)); } From f317139032d8eb26ef4ea3b7d4491f825c16e2eb Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sat, 5 Jul 2025 04:58:03 +0200 Subject: [PATCH 35/56] feat(examples): Add tile_data allocation counters --- examples/distributed/src/distributed_tile.cpp | 64 ++++++++++++------- examples/distributed/src/distributed_tile.hpp | 25 ++++++-- 2 files changed, 61 insertions(+), 28 deletions(-) diff --git a/examples/distributed/src/distributed_tile.cpp b/examples/distributed/src/distributed_tile.cpp index 50844ca3..aeabc087 100644 --- a/examples/distributed/src/distributed_tile.cpp +++ b/examples/distributed/src/distributed_tile.cpp @@ -3,23 +3,6 @@ #include #include -tile_cache::tile_cache() : - cache_(16) -{ } - -bool tile_cache::try_get(const hpx::naming::gid_type &key, tile_data &cached_data) -{ - std::lock_guard g(mutex_); - hpx::naming::gid_type unused; - return cache_.get_entry(key, unused, cached_data); -} - -void tile_cache::insert(const hpx::naming::gid_type &key, const tile_data &data) -{ - std::lock_guard g(mutex_); - cache_.insert(key, data); -} - struct tile_cache_counters { // XXX: you can do this with templates, but it's quite a bit more complicated @@ -40,6 +23,8 @@ struct tile_cache_counters }; std::atomic tile_transmission_time(0); +std::atomic tile_data_allocations(0); +std::atomic tile_data_deallocations(0); void record_transmission_time(std::int64_t elapsed_ns) { @@ -47,10 +32,16 @@ void record_transmission_time(std::int64_t elapsed_ns) tile_transmission_time += elapsed_ns; } -std::uint64_t get_transmission_time(bool reset) -{ - return hpx::util::get_and_reset_value(tile_transmission_time, reset); -} +void track_tile_data_allocation(std::size_t size) { tile_data_allocations += 1; } + +void track_tile_data_deallocation(std::size_t size) { tile_data_deallocations += 1; } + +#define GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(name) \ + std::uint64_t get_##name(bool reset) { return hpx::util::get_and_reset_value(name, reset); } + +GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_data_allocations) +GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_data_deallocations) +GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_transmission_time) void register_distributed_tile_counters() { @@ -80,10 +71,39 @@ void register_distributed_tile_counters() hpx::performance_counters::counter_type::monotonically_increasing); hpx::performance_counters::install_counter_type( "/gprat/tile_cache/transmission_time", - &get_transmission_time, + &get_tile_transmission_time, "", "", hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_data/num_allocations", + &get_tile_data_allocations, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_data/num_deallocations", + &get_tile_data_deallocations, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); +} + +tile_cache::tile_cache() : + cache_(16) +{ } + +bool tile_cache::try_get(const hpx::naming::gid_type &key, tile_data &cached_data) +{ + std::lock_guard g(mutex_); + hpx::naming::gid_type unused; + return cache_.get_entry(key, unused, cached_data); +} + +void tile_cache::insert(const hpx::naming::gid_type &key, const tile_data &data) +{ + std::lock_guard g(mutex_); + cache_.insert(key, data); } // The macros below are necessary to generate the code required for exposing diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index dba3f086..e8659686 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -11,6 +11,12 @@ #include #include +void register_distributed_tile_counters(); +void record_transmission_time(std::int64_t elapsed_ns); + +void track_tile_data_allocation(std::size_t size); +void track_tile_data_deallocation(std::size_t size); + template struct tile_data { @@ -29,9 +35,17 @@ struct tile_data }; // In case we want pooling down the road... - static T *allocate(std::size_t n) { return new T[n]; } + static T *allocate(std::size_t n) + { + track_tile_data_allocation(n); + return new T[n]; + } - static void deallocate(T *p) noexcept { delete[] p; } + static void deallocate(T *p) noexcept + { + track_tile_data_deallocation(0); // we don't know here + delete[] p; + } public: tile_data() = default; @@ -119,6 +133,7 @@ HPX_REGISTER_ACTION_DECLARATION(tile_server::set_data_action, set_data_action); class tile_cache { friend struct tile_cache_counters; + public: tile_cache(); @@ -139,9 +154,6 @@ inline tile_cache &get_tile_cache() return cache; } -void register_distributed_tile_counters(); -void record_transmission_time(std::int64_t elapsed_ns); - struct tile_handle : hpx::components::client_base { using base_type = hpx::components::client_base; @@ -191,7 +203,8 @@ struct tile_handle : hpx::components::client_base tile_server::get_data_action act; return hpx::dataflow( - [self = *this, current_gid, timer = hpx::chrono::high_resolution_timer()](hpx::future> f) mutable + [self = *this, current_gid, timer = hpx::chrono::high_resolution_timer()]( + hpx::future> f) mutable { (void) self; // need this ref-counted object to stay alive! auto data = f.get(); From 45860b54362eb683a0091fbfcb10c7466b1da6a8 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Mon, 7 Jul 2025 02:34:01 +0200 Subject: [PATCH 36/56] chore(examples): Add tile_server allocation tracking --- examples/distributed/src/distributed_tile.cpp | 24 ++++++++++++++++++- examples/distributed/src/distributed_tile.hpp | 14 ++++++++--- examples/distributed/src/main.cpp | 2 ++ 3 files changed, 36 insertions(+), 4 deletions(-) diff --git a/examples/distributed/src/distributed_tile.cpp b/examples/distributed/src/distributed_tile.cpp index aeabc087..108edd85 100644 --- a/examples/distributed/src/distributed_tile.cpp +++ b/examples/distributed/src/distributed_tile.cpp @@ -25,6 +25,8 @@ struct tile_cache_counters std::atomic tile_transmission_time(0); std::atomic tile_data_allocations(0); std::atomic tile_data_deallocations(0); +std::atomic tile_server_allocations(0); +std::atomic tile_server_deallocations(0); void record_transmission_time(std::int64_t elapsed_ns) { @@ -36,12 +38,20 @@ void track_tile_data_allocation(std::size_t size) { tile_data_allocations += 1; void track_tile_data_deallocation(std::size_t size) { tile_data_deallocations += 1; } +void track_tile_server_allocation(std::size_t size) { tile_server_allocations += 1; } + +void track_tile_server_deallocation(std::size_t size) { tile_server_deallocations += 1; } + #define GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(name) \ std::uint64_t get_##name(bool reset) { return hpx::util::get_and_reset_value(name, reset); } +GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_transmission_time) GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_data_allocations) GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_data_deallocations) -GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_transmission_time) +GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_server_allocations) +GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_server_deallocations) + +#undef GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR void register_distributed_tile_counters() { @@ -87,6 +97,18 @@ void register_distributed_tile_counters() "", "", hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_server/num_allocations", + &get_tile_server_allocations, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_server/num_deallocations", + &get_tile_server_deallocations, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); } tile_cache::tile_cache() : diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index e8659686..f0e2fc46 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -17,6 +17,9 @@ void record_transmission_time(std::int64_t elapsed_ns); void track_tile_data_allocation(std::size_t size); void track_tile_data_deallocation(std::size_t size); +void track_tile_server_allocation(std::size_t size); +void track_tile_server_deallocation(std::size_t size); + template struct tile_data { @@ -108,7 +111,14 @@ struct tile_server : hpx::components::component_base explicit tile_server(const tile_data &data) : data_(data) - { } + { + track_tile_server_allocation(data.size()); + } + + ~tile_server() + { + track_tile_server_deallocation(data_.size()); + } tile_data get_data() const { return data_; } @@ -199,8 +209,6 @@ struct tile_handle : hpx::components::client_base } } - const auto start = hpx::chrono::high_resolution_clock::now(); - tile_server::get_data_action act; return hpx::dataflow( [self = *this, current_gid, timer = hpx::chrono::high_resolution_timer()]( diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index bbf0994a..87699532 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -200,6 +200,8 @@ cholesky_hpx(Scheduler &sched, result[i * n_tiles + j] = tiles[i * n_tiles + j].get_data().get(); } } + + // hpx::get_runtime_distributed().evaluate_active_counters(false, "POST cholesky"); return result; } From 6a1e1de50c32e35611eeb3f00ccaf11d96c833c5 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Mon, 14 Jul 2025 19:20:02 +0200 Subject: [PATCH 37/56] refactor!(examples): Switch to new distributed data model --- examples/distributed/src/distributed_blas.cpp | 94 ++-- examples/distributed/src/distributed_blas.hpp | 51 ++- .../distributed/src/distributed_cholesky.hpp | 111 +++-- examples/distributed/src/distributed_tile.cpp | 77 ++-- examples/distributed/src/distributed_tile.hpp | 433 +++++++++++++----- examples/distributed/src/main.cpp | 204 +++------ examples/distributed/src/scheduling.cpp | 67 +++ examples/distributed/src/scheduling.hpp | 98 +++- test/src/output_correctness.cpp | 2 +- 9 files changed, 734 insertions(+), 403 deletions(-) create mode 100644 examples/distributed/src/scheduling.cpp diff --git a/examples/distributed/src/distributed_blas.cpp b/examples/distributed/src/distributed_blas.cpp index 9a2f0c4f..ce74602b 100644 --- a/examples/distributed/src/distributed_blas.cpp +++ b/examples/distributed/src/distributed_blas.cpp @@ -1,65 +1,73 @@ #include "distributed_blas.hpp" -#include "cpu/adapter_cblas_fp64.hpp" +#include "gprat/cpu/adapter_cblas_fp64.hpp" + #include #include -HPX_REGISTER_ACTION_DECLARATION(potrf_distributed_action); -HPX_REGISTER_ACTION_DECLARATION(trsm_distributed_action); -HPX_REGISTER_ACTION_DECLARATION(syrk_distributed_action); -HPX_REGISTER_ACTION_DECLARATION(gemm_distributed_action); +HPX_REGISTER_ACTION(GPRAT_NS::potrf_distributed_action); +HPX_REGISTER_ACTION(GPRAT_NS::trsm_distributed_action); +HPX_REGISTER_ACTION(GPRAT_NS::syrk_distributed_action); +HPX_REGISTER_ACTION(GPRAT_NS::gemm_distributed_action); + +GPRAT_NS_BEGIN -GPRAT_DEFINE_PLAIN_ACTION_FOR(&inplace::potrf); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&inplace::trsm); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&inplace::syrk); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&inplace::gemm); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&potrf); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&trsm); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&syrk); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&gemm); -tile_handle potrf_distributed(const tile_handle &A, int N) +hpx::future> potrf_distributed(const tile_handle &A, int N) { return hpx::dataflow( hpx::launch::async, hpx::unwrapping( - [A, N](tile_data tile) + [A, N](const mutable_tile_data &tile) { - inplace::potrf(tile, N); - return tile_handle(hpx::colocated(A.get_id()), tile); + GPRAT_TIME_PLAIN_ACTION(potrf); + return A.set_async(potrf(tile, N)); }), - A.get_data()); + A.get_async()); } -tile_handle -trsm_distributed(const tile_handle &L, const tile_handle &A, int N, int M, BLAS_TRANSPOSE transpose_L, BLAS_SIDE side_L) +hpx::future> trsm_distributed( + const tile_handle &L, + const tile_handle &A, + int N, + int M, + BLAS_TRANSPOSE transpose_L, + BLAS_SIDE side_L) { return hpx::dataflow( hpx::launch::async, hpx::unwrapping( - [L, A, N, M, transpose_L, side_L](const tile_data &Ld, tile_data Ad) + [A, N, M, transpose_L, side_L](const mutable_tile_data &Ld, mutable_tile_data Ad) { - inplace::trsm(Ld, Ad, N, M, transpose_L, side_L); - return tile_handle(hpx::colocated(A.get_id()), Ad); + GPRAT_TIME_PLAIN_ACTION(trsm); + return A.set_async(trsm(Ld, Ad, N, M, transpose_L, side_L)); }), - L.get_data(), - A.get_data()); + L.get_async(), + A.get_async()); } -tile_handle syrk_distributed(const tile_handle &A, const tile_handle &B, int N) +hpx::future> syrk_distributed(const tile_handle &A, const tile_handle &B, int N) { return hpx::dataflow( hpx::launch::async, hpx::unwrapping( - [A, B, N](tile_data Ad, const tile_data &Bd) + [A, N](mutable_tile_data Ad, const mutable_tile_data &Bd) { - inplace::syrk(Ad, Bd, N); - return tile_handle(hpx::colocated(A.get_id()), Ad); + GPRAT_TIME_PLAIN_ACTION(syrk); + return A.set_async(syrk(Ad, Bd, N)); }), - A.get_data(), - B.get_data()); + A.get_async(), + B.get_async()); } -tile_handle gemm_distributed( - const tile_handle &A, - const tile_handle &B, - const tile_handle &C, +hpx::future> gemm_distributed( + const tile_handle &A, + const tile_handle &B, + const tile_handle &C, int N, int M, int K, @@ -69,41 +77,43 @@ tile_handle gemm_distributed( return hpx::dataflow( hpx::launch::async, hpx::unwrapping( - [A, B, C, N, M, K, transpose_A, transpose_B]( - const tile_data &Ad, const tile_data &Bd, tile_data Cd) + [C, N, M, K, transpose_A, transpose_B]( + const mutable_tile_data &Ad, const mutable_tile_data &Bd, mutable_tile_data Cd) { - inplace::gemm(Ad, Bd, Cd, N, M, K, transpose_A, transpose_B); - return tile_handle(hpx::colocated(C.get_id()), Cd); + GPRAT_TIME_PLAIN_ACTION(gemm); + return C.set_async(gemm(Ad, Bd, Cd, N, M, K, transpose_A, transpose_B)); }), - A.get_data(), - B.get_data(), - C.get_data()); + A.get_async(), + B.get_async(), + C.get_async()); } void register_distributed_blas_counters() { hpx::performance_counters::install_counter_type( "/gprat/potrf/time", - get_and_reset_plain_action_elapsed<&inplace::potrf>, + get_and_reset_plain_action_elapsed<&potrf>, "", "", hpx::performance_counters::counter_type::monotonically_increasing); hpx::performance_counters::install_counter_type( "/gprat/trsm/time", - get_and_reset_plain_action_elapsed<&inplace::trsm>, + get_and_reset_plain_action_elapsed<&trsm>, "", "", hpx::performance_counters::counter_type::monotonically_increasing); hpx::performance_counters::install_counter_type( "/gprat/syrk/time", - get_and_reset_plain_action_elapsed<&inplace::syrk>, + get_and_reset_plain_action_elapsed<&syrk>, "", "", hpx::performance_counters::counter_type::monotonically_increasing); hpx::performance_counters::install_counter_type( "/gprat/gemm/time", - get_and_reset_plain_action_elapsed<&inplace::gemm>, + get_and_reset_plain_action_elapsed<&gemm>, "", "", hpx::performance_counters::counter_type::monotonically_increasing); } + +GPRAT_NS_END diff --git a/examples/distributed/src/distributed_blas.hpp b/examples/distributed/src/distributed_blas.hpp index 867aeac7..493cf548 100644 --- a/examples/distributed/src/distributed_blas.hpp +++ b/examples/distributed/src/distributed_blas.hpp @@ -1,32 +1,49 @@ #pragma once -#include "cpu/adapter_cblas_fp64.hpp" +#include "gprat/cpu/adapter_cblas_fp64.hpp" + #include "distributed_tile.hpp" #include "scheduling.hpp" #include -tile_handle potrf_distributed(const tile_handle &A, int N); -tile_handle trsm_distributed( - const tile_handle &L, const tile_handle &A, int N, int M, BLAS_TRANSPOSE transpose_L, BLAS_SIDE side_L); -tile_handle syrk_distributed(const tile_handle &A, const tile_handle &B, int N); -tile_handle gemm_distributed( - const tile_handle &A, - const tile_handle &B, - const tile_handle &C, +GPRAT_REGISTER_TILED_DATASET_DECLARATION(double, double); + +GPRAT_NS_BEGIN + +hpx::future> potrf_distributed(const tile_handle &A, int N); +hpx::future> trsm_distributed( + const tile_handle &L, + const tile_handle &A, + int N, + int M, + BLAS_TRANSPOSE transpose_L, + BLAS_SIDE side_L); +hpx::future> syrk_distributed(const tile_handle &A, const tile_handle &B, int N); +hpx::future> gemm_distributed( + const tile_handle &A, + const tile_handle &B, + const tile_handle &C, int N, int M, int K, BLAS_TRANSPOSE transpose_A, BLAS_TRANSPOSE transpose_B); -HPX_DEFINE_PLAIN_ACTION(potrf_distributed); -HPX_DEFINE_PLAIN_ACTION(trsm_distributed); -HPX_DEFINE_PLAIN_ACTION(syrk_distributed); -HPX_DEFINE_PLAIN_ACTION(gemm_distributed); +HPX_DEFINE_PLAIN_DIRECT_ACTION(potrf_distributed); +HPX_DEFINE_PLAIN_DIRECT_ACTION(trsm_distributed); +HPX_DEFINE_PLAIN_DIRECT_ACTION(syrk_distributed); +HPX_DEFINE_PLAIN_DIRECT_ACTION(gemm_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&inplace::potrf, potrf_distributed_action, "POTRF"); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&inplace::trsm, trsm_distributed_action, "TRSM"); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&inplace::syrk, syrk_distributed_action, "SYRK"); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&inplace::gemm, gemm_distributed_action, "GEMM"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&potrf, potrf_distributed_action, "POTRF"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&trsm, trsm_distributed_action, "TRSM"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&syrk, syrk_distributed_action, "SYRK"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&gemm, gemm_distributed_action, "GEMM"); void register_distributed_blas_counters(); + +GPRAT_NS_END + +HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::potrf_distributed_action); +HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::trsm_distributed_action); +HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::syrk_distributed_action); +HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::gemm_distributed_action); diff --git a/examples/distributed/src/distributed_cholesky.hpp b/examples/distributed/src/distributed_cholesky.hpp index 098f97b1..2f34e4eb 100644 --- a/examples/distributed/src/distributed_cholesky.hpp +++ b/examples/distributed/src/distributed_cholesky.hpp @@ -4,56 +4,99 @@ #include "scheduling.hpp" #include -struct tiled_cholesky_distribution_policy_paap12 +GPRAT_NS_BEGIN + +template +std::vector>> +make_cholesky_dataset(const tiled_scheduler_local &, std::size_t num_tiles) { - constexpr std::size_t locality_for_tile(std::size_t row, std::size_t col) const - { - return (row + col) % num_localities; - } + return { num_tiles * num_tiles }; +} - constexpr std::size_t locality_for_POTRF(std::size_t k) const { return (2 * k) % num_localities; } +// Default implementations in case the scheduler provides none +constexpr std::size_t cholesky_tile(...) { return 0; } - constexpr std::size_t locality_for_SYRK(std::size_t m) const { return (2 * m) % num_localities; } +constexpr std::size_t cholesky_POTRF(...) { return 0; } - constexpr std::size_t locality_for_TRSM(std::size_t k, std::size_t m) const { return (k + m) % num_localities; } +constexpr std::size_t cholesky_SYRK(...) { return 0; } - constexpr std::size_t locality_for_GEMM(std::size_t /*k*/, std::size_t m, std::size_t n) const - { - return (m + n) % num_localities; - } +constexpr std::size_t cholesky_TRSM(...) { return 0; } - std::size_t num_localities; -}; +constexpr std::size_t cholesky_GEMM(...) { return 0; } + +constexpr std::size_t cholesky_TRSV(...) { return 0; } + +constexpr std::size_t cholesky_GEMV(...) { return 0; } -template -struct tiled_cholesky_scheduler_distributed +namespace scheduler { - using tiled_matrix_handles = std::vector; - tiled_cholesky_scheduler_distributed() = default; +struct tiled_cholesky_scheduler_paap12 : tiled_scheduler_distributed +{ + using tiled_scheduler_distributed::tiled_scheduler_distributed; + + std::size_t num_localities = localities_.size(); +}; + +template +tiled_dataset make_cholesky_dataset(const tiled_cholesky_scheduler_paap12 &policy, std::size_t num_tiles) +{ + std::vector> targets; + targets.reserve(policy.num_localities); - [[nodiscard]] schedule_on_locality for_tile(std::size_t row, std::size_t col) const + for (std::size_t i = 0; i < policy.num_localities; ++i) { - return localities[policy.locality_for_tile(row, col)]; + targets.emplace_back(policy.localities_[i], 0); } - [[nodiscard]] schedule_on_locality for_POTRF(std::size_t k) const + for (std::size_t row = 0; row < num_tiles; row++) { - return localities[policy.locality_for_POTRF(k)]; + for (std::size_t col = 0; col < num_tiles; col++) + { + const auto l = (row + col) % policy.num_localities; + ++targets[l].second; + } } - [[nodiscard]] schedule_on_locality for_SYRK(std::size_t m) const { return localities[policy.locality_for_SYRK(m)]; } + return tiled_dataset_accessor{ targets, num_tiles * num_tiles }.to_dataset(); +} - [[nodiscard]] schedule_on_locality for_TRSM(std::size_t k, std::size_t m) const - { - return localities[policy.locality_for_TRSM(k, m)]; - } +constexpr std::size_t cholesky_tile(const tiled_cholesky_scheduler_paap12 &policy, std::size_t row, std::size_t col) +{ + return (row + col) % policy.num_localities; +} - [[nodiscard]] schedule_on_locality for_GEMM(std::size_t k, std::size_t m, std::size_t n) const - { - return localities[policy.locality_for_GEMM(k, m, n)]; - } +constexpr std::size_t cholesky_POTRF(const tiled_cholesky_scheduler_paap12 &policy, std::size_t k) +{ + return (2 * k) % policy.num_localities; +} - std::vector localities = hpx::find_all_localities(); - DistPolicy policy{ localities.size() }; -}; +constexpr std::size_t cholesky_SYRK(const tiled_cholesky_scheduler_paap12 &policy, std::size_t m) +{ + return (2 * m) % policy.num_localities; +} + +constexpr std::size_t cholesky_TRSM(const tiled_cholesky_scheduler_paap12 &policy, std::size_t k, std::size_t m) +{ + return (k + m) % policy.num_localities; +} + +constexpr std::size_t +cholesky_GEMM(const tiled_cholesky_scheduler_paap12 &policy, std::size_t /*k*/, std::size_t m, std::size_t n) +{ + return (m + n) % policy.num_localities; +} + +constexpr std::size_t cholesky_TRSV(const tiled_cholesky_scheduler_paap12 &policy, std::size_t k) +{ + return k % policy.num_localities; +} + +constexpr std::size_t cholesky_GEMV(const tiled_cholesky_scheduler_paap12 &policy, std::size_t k) +{ + return k % policy.num_localities; +} + +} // namespace scheduler + +GPRAT_NS_END diff --git a/examples/distributed/src/distributed_tile.cpp b/examples/distributed/src/distributed_tile.cpp index 108edd85..dbc122a5 100644 --- a/examples/distributed/src/distributed_tile.cpp +++ b/examples/distributed/src/distributed_tile.cpp @@ -3,6 +3,11 @@ #include #include +HPX_DISTRIBUTED_METADATA(GPRAT_NS::server::tiled_dataset_config_data, gprat_server_tiled_dataset_config_data) + +GPRAT_NS_BEGIN + +/* struct tile_cache_counters { // XXX: you can do this with templates, but it's quite a bit more complicated @@ -21,7 +26,7 @@ struct tile_cache_counters #undef GPRAT_MAKE_STATISTICS_ACCESSOR }; - +*/ std::atomic tile_transmission_time(0); std::atomic tile_data_allocations(0); std::atomic tile_data_deallocations(0); @@ -54,31 +59,32 @@ GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_server_deallocations) #undef GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR void register_distributed_tile_counters() -{ - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/hits", - &tile_cache_counters::get_cache_hits, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/misses", - &tile_cache_counters::get_cache_misses, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/evictions", - &tile_cache_counters::get_cache_evictions, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/insertions", - &tile_cache_counters::get_cache_insertions, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); +{ /* + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/hits", + &tile_cache_counters::get_cache_hits, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/misses", + &tile_cache_counters::get_cache_misses, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/evictions", + &tile_cache_counters::get_cache_evictions, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/insertions", + &tile_cache_counters::get_cache_insertions, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + */ hpx::performance_counters::install_counter_type( "/gprat/tile_cache/transmission_time", &get_tile_transmission_time, @@ -111,6 +117,7 @@ void register_distributed_tile_counters() hpx::performance_counters::counter_type::monotonically_increasing); } +/* tile_cache::tile_cache() : cache_(16) { } @@ -126,20 +133,6 @@ void tile_cache::insert(const hpx::naming::gid_type &key, const tile_data(). -typedef hpx::components::component tile_server_type; -HPX_REGISTER_COMPONENT(tile_server_type, tile_server) - -// HPX_REGISTER_ACTION() exposes the component member function for remote -// invocation. -typedef tile_server::get_data_action get_data_action; -HPX_REGISTER_ACTION(get_data_action) - -typedef tile_server::set_data_action set_data_action; -HPX_REGISTER_ACTION(set_data_action) +GPRAT_NS_END diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index f0e2fc46..74c6a93a 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -1,5 +1,7 @@ #pragma once +#include "gprat/tile_data.hpp" + #include #include #include @@ -8,8 +10,12 @@ #include #include #include +#include #include #include +#include + +GPRAT_NS_BEGIN void register_distributed_tile_counters(); void record_transmission_time(std::int64_t elapsed_ns); @@ -20,165 +26,358 @@ void track_tile_data_deallocation(std::size_t size); void track_tile_server_allocation(std::size_t size); void track_tile_server_deallocation(std::size_t size); -template -struct tile_data +namespace server { - private: - typedef hpx::serialization::serialize_buffer buffer_type; - - struct hold_reference +struct tiled_dataset_config_data +{ + struct tile_entry { - explicit hold_reference(const buffer_type &data) : - data_(data) + tile_entry() : + locality_id(hpx::naming::invalid_locality_id), + generation(0) + { } + + tile_entry(hpx::id_type tile, std::uint32_t locality_id, std::uint64_t generation) : + tile(std::move(tile)), + locality_id(locality_id), + generation(generation) { } - void operator()(const double *) const { } // no deletion necessary + hpx::id_type tile; + std::uint32_t locality_id; + std::uint64_t generation; + + private: + friend class hpx::serialization::access; - buffer_type data_; + template + void serialize(Archive &ar, unsigned) + { + ar & tile & locality_id & generation; + } }; - // In case we want pooling down the road... - static T *allocate(std::size_t n) + tiled_dataset_config_data() = default; + + tiled_dataset_config_data(std::vector &&tiles) : + tiles(std::move(tiles)) + { } + + std::vector tiles; + + private: + friend class hpx::serialization::access; + + template + void serialize(Archive &ar, unsigned) { - track_tile_data_allocation(n); - return new T[n]; + ar & tiles; } +}; +} // namespace server + +GPRAT_NS_END + +HPX_DISTRIBUTED_METADATA_DECLARATION(GPRAT_NS::server::tiled_dataset_config_data, + gprat_server_tiled_dataset_config_data) + +GPRAT_NS_BEGIN + +namespace server +{ +// This is the server side representation of the data. We expose this as a HPX +// component which allows for it to be created and accessed remotely through +// a global address (hpx::id_type). +template +struct tile_server : hpx::components::locking_hook>> +{ + tile_server() = default; - static void deallocate(T *p) noexcept + explicit tile_server(const mutable_tile_data &data) : + data_(data) { - track_tile_data_deallocation(0); // we don't know here - delete[] p; + track_tile_server_allocation(data.size()); } - public: - tile_data() = default; + ~tile_server() { track_tile_server_deallocation(data_.size()); } - // Create a new (uninitialized) partition of the given size. - explicit tile_data(std::size_t size) : - data_(allocate(size), size, buffer_type::take, &tile_data::deallocate) - { } + [[nodiscard]] mutable_tile_data get_data() const { return data_; } - // Create a partition which acts as a proxy to a part of the embedded array. - // The proxy is assumed to refer to either the left or the right boundary - // element. - tile_data(const tile_data &base, std::size_t offset, std::size_t size) : - data_(base.data_.data() + offset, - size, - buffer_type::reference, - hold_reference(base.data_)) // keep referenced partition alive - { } + void set_data(const mutable_tile_data &data) { data_ = data; } - [[nodiscard]] T *data() noexcept { return data_.data(); } + // Every member function that has to be invoked remotely needs to be + // wrapped into a component action. + HPX_DEFINE_COMPONENT_DIRECT_ACTION(tile_server, get_data) + HPX_DEFINE_COMPONENT_DIRECT_ACTION(tile_server, set_data) + + private: + mutable_tile_data data_; +}; + +} // namespace server + +#define GPRAT_REGISTER_TILED_DATASET_DECLARATION_IMPL(type, name) \ + HPX_REGISTER_ACTION_DECLARATION(type::get_data_action, HPX_PP_CAT(_tiled_dataset_get_data_action_, name)) \ + HPX_REGISTER_ACTION_DECLARATION(type::set_data_action, HPX_PP_CAT(_tiled_dataset_set_data_action_, name)) \ + /**/ + +#define GPRAT_REGISTER_TILED_DATASET_DECLARATION(type, name) \ + typedef ::GPRAT_NS::server::tile_server HPX_PP_CAT(_tiled_dataset_server_, HPX_PP_CAT(type, name)); \ + GPRAT_REGISTER_TILED_DATASET_DECLARATION_IMPL(HPX_PP_CAT(_tiled_dataset_server_, HPX_PP_CAT(type, name)), name) + +#define GPRAT_REGISTER_TILED_DATASET_IMPL(type, name) \ + HPX_REGISTER_ACTION(type::get_data_action, HPX_PP_CAT(_tiled_dataset_get_data_action_, name)) \ + HPX_REGISTER_ACTION(type::set_data_action, HPX_PP_CAT(_tiled_dataset_set_data_action_, name)) \ + typedef ::hpx::components::component HPX_PP_CAT(_tiled_dataset_server_component_, name); \ + HPX_REGISTER_COMPONENT(HPX_PP_CAT(_tiled_dataset_server_component_, name)) \ + /**/ - [[nodiscard]] const T *data() const noexcept { return data_.data(); } +#define GPRAT_REGISTER_TILED_DATASET(type, name) \ + typedef ::GPRAT_NS::server::tile_server HPX_PP_CAT(_tiled_dataset_server_, HPX_PP_CAT(type, name)); \ + GPRAT_REGISTER_TILED_DATASET_IMPL(HPX_PP_CAT(_tiled_dataset_server_, HPX_PP_CAT(type, name)), name) - [[nodiscard]] std::size_t size() const noexcept { return data_.size(); } +template +class tiled_dataset_accessor; - // ReSharper disable once CppNonExplicitConversionOperator - operator std::span() noexcept { return { data_.data(), data_.size() }; } // NOLINT(*-explicit-constructor) +template +class tile_handle +{ + public: + tile_handle() = default; - // ReSharper disable once CppNonExplicitConversionOperator - operator std::span() const noexcept // NOLINT(*-explicit-constructor) + tile_handle(const hpx::id_type &id, std::size_t tile_index, std::size_t generation) : + ds_(id), + tile_index_(tile_index), + generation_(generation) + { } + + operator mutable_tile_data() const { return get(); } + + mutable_tile_data get() const { - return { data_.data(), data_.size() }; + tiled_dataset_accessor ds(ds_); // TRANSITION + return ds.get_tile_data(tile_index_, generation_).get(); + } + + hpx::future> get_async() const + { + tiled_dataset_accessor ds(ds_); // TRANSITION + return ds.get_tile_data(tile_index_, generation_); + } + + hpx::future set_async(const mutable_tile_data &data) const + { + tiled_dataset_accessor ds(ds_); // TRANSITION + return ds.set_tile_data(tile_index_, generation_ + 1, data); } private: - // Serialization support: even if all of the code below runs on one - // locality only, we need to provide an (empty) implementation for the - // serialization as all arguments passed to actions have to support this. friend class hpx::serialization::access; template - void serialize(Archive &ar, const unsigned int) + void serialize(Archive &ar, unsigned) { - // clang-format off - ar & data_; - // clang-format on + ar & ds_ & tile_index_ & generation_; } - buffer_type data_; + // we need this instead of the actual tile_handle because per-locality caches + // reside in the accessor. + hpx::id_type ds_; + std::size_t tile_index_; + std::size_t generation_; }; -/////////////////////////////////////////////////////////////////////////////// -// This is the server side representation of the data. We expose this as a HPX -// component which allows for it to be created and accessed remotely through -// a global address (hpx::id_type). -struct tile_server : hpx::components::component_base +template +using tiled_dataset = std::vector>>; + +template +class tiled_dataset_accessor + : public hpx::components::client_base< + tiled_dataset_accessor, + hpx::components::server::distributed_metadata_base> { - // construct new instances - tile_server() = default; + using server_type = hpx::components::server::distributed_metadata_base; + using base_type = hpx::components::client_base, server_type>; - explicit tile_server(const tile_data &data) : - data_(data) + using tile_server = server::tile_server; + + struct tile_entry : server::tiled_dataset_config_data::tile_entry { - track_tile_server_allocation(data.size()); + using base_type = server::tiled_dataset_config_data::tile_entry; + + tile_entry() = default; + + tile_entry(const hpx::id_type &part, std::uint32_t locality_id, std::size_t version) : + base_type(part, locality_id, version) + { } + + tile_entry(const base_type &base) noexcept : + base_type(base) + { } + + tile_entry(base_type &&base) noexcept : + base_type(HPX_MOVE(base)) + { } + + std::shared_ptr local_data; + }; + + // The list of partitions belonging to this vector. + // Each partition is described by its corresponding client object, its + // size, and locality id. + using tiles_vector_type = std::vector; + + public: + explicit tiled_dataset_accessor(const hpx::id_type &id) { connect_to(id).get(); } + + explicit tiled_dataset_accessor(std::span> targets, + std::size_t num_tiles) + { + create(targets, num_tiles); } - ~tile_server() + hpx::future connect_to(const hpx::id_type &id) { - track_tile_server_deallocation(data_.size()); + return hpx::async(server_type::get_action(), id) + .then([this, id](hpx::future &&f) -> void + { return assign_existing(id, f.get()); }); } - tile_data get_data() const { return data_; } + hpx::future> get_tile_data(std::size_t tile_index, std::size_t /*generation*/) const + { + if (tiles_[tile_index].local_data) + { + return hpx::make_ready_future(tiles_[tile_index].local_data->get_data()); + } - void set_data(const tile_data &data) { data_ = data; } + typename server::tile_server::get_data_action act; + return hpx::async(act, tiles_[tile_index].tile); + } - // Every member function that has to be invoked remotely needs to be - // wrapped into a component action. - HPX_DEFINE_COMPONENT_DIRECT_ACTION(tile_server, get_data, get_data_action) - HPX_DEFINE_COMPONENT_DIRECT_ACTION(tile_server, set_data, set_data_action) + hpx::future> + set_tile_data(std::size_t tile_index, std::size_t generation, const mutable_tile_data &data) const + { + if (tiles_[tile_index].local_data) + { + tiles_[tile_index].local_data->set_data(data); + return hpx::make_ready_future(tile_handle{ base_type::get(), tile_index, generation + 1 }); + } + + typename server::tile_server::set_data_action act; + return hpx::async(act, tiles_[tile_index].tile, data) + .then([this, tile_index, generation](const hpx::future &) + { return tile_handle{ base_type::get(), tile_index, generation + 1 }; }); + } + + // TRANSITION + tiled_dataset to_dataset() + { + tiled_dataset result; + result.reserve(tiles_.size()); + for (std::size_t i = 0; i < tiles_.size(); ++i) + { + result.emplace_back(hpx::make_ready_future(tile_handle{ base_type::get(), i, tiles_[i].generation })); + } + return result; + } private: - tile_data data_; -}; + void assign_existing(const hpx::id_type &id, server::tiled_dataset_config_data &&config) + { + tiles_.clear(); + tiles_.insert(tiles_.end(), config.tiles.begin(), config.tiles.end()); + + const auto here = hpx::get_locality_id(); + for (auto &tile : tiles_) + { + if (tile.locality_id == here && !tile.local_data) + { + tile.local_data = hpx::get_ptr(hpx::launch::sync, tile.tile); + } + } -HPX_REGISTER_ACTION_DECLARATION(tile_server::get_data_action, get_data_action); -HPX_REGISTER_ACTION_DECLARATION(tile_server::set_data_action, set_data_action); + return base_type::reset(id); + } -/////////////////////////////////////////////////////////////////////////////// -// This is a client side helper class allowing to hide some of the tedious -// boilerplate while referencing a remote partition. + void create(std::span> targets, std::size_t num_tiles) + { + std::vector>> objs; + objs.reserve(targets.size()); + for (const auto &target : targets) + { + objs.emplace_back(hpx::components::bulk_create_async(target.first, target.second)); + } -class tile_cache -{ - friend struct tile_cache_counters; + const auto here = hpx::get_locality_id(); + tiles_.resize(num_tiles); - public: - tile_cache(); + std::size_t l = 0; + for (std::size_t i = 0; i < targets.size(); ++i) + { + const auto locality = hpx::naming::get_locality_id_from_id(targets[i].first); + for (const hpx::id_type &id : objs[i].get()) + { + tiles_[l] = tile_entry(id, locality, 0); - bool try_get(const hpx::naming::gid_type &key, tile_data &cached_data); + if (locality == here) + { + tiles_[l].local_data = hpx::get_ptr(hpx::launch::sync, id); + } - void insert(const hpx::naming::gid_type &key, const tile_data &data); + if (++l == num_tiles) + { + break; + } + } + } + HPX_ASSERT(l == num_tiles); - private: - hpx::mutex mutex_; - hpx::util::cache:: - lru_cache, hpx::util::cache::statistics::local_full_statistics> - cache_; + std::vector data{ tiles_.begin(), tiles_.end() }; + base_type::reset( + hpx::new_>( + hpx::find_here(), server::tiled_dataset_config_data{ std::move(data) })); + } + + tiles_vector_type tiles_; }; -inline tile_cache &get_tile_cache() +// partitioned_vector_partition => tile_handle? +// server::partitioned_vector => tile_server? + +/*template +class tiled_dataset { - static tile_cache cache; - return cache; -} -struct tile_handle : hpx::components::client_base + + hpx::future>& operator[](std::size_t index) + { + return + } + // operator[] => future[tile_reference]& +};*/ + +// OLD: +/* +/////////////////////////////////////////////////////////////////////////////// +// This is a client side helper class allowing to hide some of the tedious +// boilerplate while referencing a remote partition. + +template +struct tile_handle : hpx::components::client_base, server::tile_server> { - using base_type = hpx::components::client_base; + using base_type = hpx::components::client_base>; tile_handle() = default; // Create new component on locality 'where' and initialize the held data - tile_handle(const hpx::id_type &where, const tile_data &data) : - base_type(hpx::new_(where, data)) + tile_handle(const hpx::id_type &where, const mutable_tile_data &data) : + base_type(hpx::new_>(where, data)) { } // Create new component on locality 'where' and initialize the held data template - requires hpx::traits::is_distribution_policy_v tile_handle(const T &policy, const tile_data &data) : - base_type(hpx::new_(policy, data)) + requires hpx::traits::is_distribution_policy_v tile_handle(const T &policy, const mutable_tile_data &data) : + base_type(hpx::new_>(policy, data)) { } // Attach a future representing a (possibly remote) partition. @@ -196,36 +395,26 @@ struct tile_handle : hpx::components::client_base /////////////////////////////////////////////////////////////////////////// // tile's are immutable for now, meaning we can cache them on first use. - // TODO: profile whether this strategy (new components on mutation) beats mutable tile state. - [[nodiscard]] hpx::future> get_data() const + [[nodiscard]] hpx::future> get_data() const { - hpx::naming::gid_type current_gid = get().get_gid(); - { - auto &cache = get_tile_cache(); - tile_data cached_data; - if (cache.try_get(current_gid, cached_data)) - { - return hpx::make_ready_future(cached_data); - } - } - - tile_server::get_data_action act; - return hpx::dataflow( - [self = *this, current_gid, timer = hpx::chrono::high_resolution_timer()]( - hpx::future> f) mutable - { - (void) self; // need this ref-counted object to stay alive! - auto data = f.get(); - record_transmission_time(timer.elapsed_nanoseconds()); - get_tile_cache().insert(current_gid, data); - return std::move(data); - }, - hpx::async(act, get_id())); + typename server::tile_server::get_data_action act; + return hpx::async(act, base_type::get_id()); } - [[nodiscard]] hpx::future set_data(const tile_data &data) + [[nodiscard]] hpx::future set_data(const mutable_tile_data &data) { - tile_server::set_data_action act; - return hpx::async(act, get_id(), data); + typename server::tile_server::set_data_action act; + return hpx::async(act, base_type::get_id(), data); } }; +*/ + +GPRAT_NS_END + +/* +// serialization of partitioned_vector requires special handling +template +struct hpx::traits::needs_reference_semantics> + : std::true_type +{ +};*/ diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index 87699532..52374a80 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -1,13 +1,12 @@ +#include "gprat/cpu/gp_algorithms.hpp" +#include "gprat/kernels.hpp" + #include "../../test/src/test_data.hpp" #include "distributed_blas.hpp" #include "distributed_cholesky.hpp" #include "distributed_tile.hpp" -#include "cpu/gp_functions.hpp" -#include "gp_kernels.hpp" -#include "gprat_c.hpp" -#include "cpu/tiled_algorithms.hpp" -#include "utils_c.hpp" #include +#include #include #include #include @@ -15,121 +14,48 @@ // This is a standalone test, so including this directly is fine. // Better than having the whole project depend on compiled Boost.Json! -#include - -namespace gprat_hyper -{ -template -inline void save_construct_data(Archive &ar, const SEKParams *v, const unsigned int) -{ - ar << v->lengthscale; - ar << v->vertical_lengthscale; - ar << v->noise_variance; -} +#include "gprat/gprat.hpp" +#include "gprat/utils.hpp" -template -inline void load_construct_data(Archive &ar, SEKParams *v, const unsigned int) -{ - double lengthscale, vertical_lengthscale, noise_variance; - ar >> lengthscale; - ar >> vertical_lengthscale; - ar >> noise_variance; - - // ::new(ptr) construct new object at given address - hpx::construct_at(v, lengthscale, vertical_lengthscale, noise_variance); -} - -template -void serialize(Archive &ar, SEKParams &pt, const unsigned int) -{ - ar & pt.m_T & pt.w_T; -} - -} // namespace gprat_hyper - -///////////////////////////////////////////////////////// -// Tile generation -double compute_covariance_function(std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, - std::span i_input, - std::span j_input) -{ - // k(z_i,z_j) = vertical_lengthscale * exp(-0.5 / lengthscale^2 * (z_i - z_j)^2) - double distance = 0.0; - for (std::size_t k = 0; k < n_regressors; k++) - { - const double z_ik_minus_z_jk = i_input[k] - j_input[k]; - distance += z_ik_minus_z_jk * z_ik_minus_z_jk; - } - - return sek_params.vertical_lengthscale * exp(-0.5 / (sek_params.lengthscale * sek_params.lengthscale) * distance); -} - -tile_data make_covariance_tile( - std::size_t row, - std::size_t col, - std::size_t N, - std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, - std::span input) -{ - tile_data tile(N * N); - for (std::size_t i = 0; i < N; i++) - { - std::size_t i_global = N * row + i; - for (std::size_t j = 0; j < N; j++) - { - std::size_t j_global = N * col + j; +#include - // compute covariance function - auto covariance_function = compute_covariance_function( - n_regressors, sek_params, input.subspan(i_global, n_regressors), input.subspan(j_global, n_regressors)); - if (i_global == j_global) - { - // noise variance on diagonal - covariance_function += sek_params.noise_variance; - } +GPRAT_REGISTER_TILED_DATASET(double, double); - tile.data()[i * N + j] = covariance_function; - } - } - return tile; -} +GPRAT_NS_BEGIN -tile_handle make_covariance_tile_distributed( +hpx::future> gen_tile_covariance_distributed( + tile_handle tile, std::size_t row, std::size_t col, std::size_t N, std::size_t n_regressors, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, std::span input) { - return tile_handle(hpx::find_here(), make_covariance_tile(row, col, N, n_regressors, sek_params, input)); + return tile.set_async(cpu::gen_tile_covariance(row, col, N, n_regressors, sek_params, input)); } -HPX_PLAIN_ACTION(make_covariance_tile_distributed, make_covariance_tile_action) +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_covariance_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_covariance, + gen_tile_covariance_distributed_action, + "gen_tile_covariance"); -template <> -struct plain_action_for<&make_covariance_tile> -{ - using action_type = make_covariance_tile_action; - constexpr static std::string_view name = "gen_tile_covariance"; -}; - -template > -void right_looking_cholesky_tiled( - Scheduler &sched, typename Scheduler::tiled_matrix_handles &ft_tiles, std::size_t N, std::size_t n_tiles) +template +void right_looking_cholesky_tiled(Scheduler &sched, Tiles &ft_tiles, std::size_t N, std::size_t n_tiles) { for (std::size_t k = 0; k < n_tiles; k++) { // POTRF: Compute Cholesky factor L - ft_tiles[k * n_tiles + k] = dataflow(sched.for_POTRF(k), ft_tiles[k * n_tiles + k], N); + ft_tiles[k * n_tiles + k] = detail::named_dataflow( + sched, cholesky_POTRF(sched, k), "cholesky_tiled", ft_tiles[k * n_tiles + k], N); for (std::size_t m = k + 1; m < n_tiles; m++) { // TRSM: Solve X * L^T = A - ft_tiles[m * n_tiles + k] = dataflow( - sched.for_TRSM(k, m), + ft_tiles[m * n_tiles + k] = detail::named_dataflow( + sched, + cholesky_TRSM(sched, k, m), + "cholesky_tiled", ft_tiles[k * n_tiles + k], ft_tiles[m * n_tiles + k], N, @@ -140,13 +66,20 @@ void right_looking_cholesky_tiled( for (std::size_t m = k + 1; m < n_tiles; m++) { // SYRK: A = A - B * B^T - ft_tiles[m * n_tiles + m] = - dataflow(sched.for_SYRK(m), ft_tiles[m * n_tiles + m], ft_tiles[m * n_tiles + k], N); + ft_tiles[m * n_tiles + m] = detail::named_dataflow( + sched, + cholesky_SYRK(sched, m), + "cholesky_tiled", + ft_tiles[m * n_tiles + m], + ft_tiles[m * n_tiles + k], + N); for (std::size_t n = k + 1; n < m; n++) { // GEMM: C = C - A * B^T - ft_tiles[m * n_tiles + n] = dataflow( - sched.for_GEMM(k, m, n), + ft_tiles[m * n_tiles + n] = detail::named_dataflow( + sched, + cholesky_GEMM(sched, k, m, n), + "cholesky_tiled", ft_tiles[m * n_tiles + k], ft_tiles[n * n_tiles + k], ft_tiles[m * n_tiles + n], @@ -160,29 +93,32 @@ void right_looking_cholesky_tiled( } } -template > -std::vector> +template +std::vector> cholesky_hpx(Scheduler &sched, std::span training_input, - const gprat_hyper::SEKParams &sek_params, + const SEKParams &sek_params, std::size_t n_tiles, std::size_t n_tile_size, std::size_t n_regressors) { - typename Scheduler::tiled_matrix_handles tiles(n_tiles * n_tiles); // Tiled covariance matrix - - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous assembly - // std::vector> tile_objs; - // tile_objs.reserve(n_tiles * n_tiles); + auto tiles = make_cholesky_dataset(sched, n_tiles); // Tiled covariance matrix for (std::size_t row = 0; row < n_tiles; row++) { for (std::size_t col = 0; col <= row; col++) { - tiles[row * n_tiles + col] = - tile_handle(sched.for_tile(row, col).where, - make_covariance_tile(row, col, n_tile_size, n_regressors, sek_params, training_input)); + tiles[row * n_tiles + col] = detail::named_dataflow( + sched, + cholesky_tile(sched, row, col), + "cholesky init", + tiles[row * n_tiles + col], + row, + col, + n_tile_size, + n_regressors, + sek_params, + training_input); } } @@ -192,15 +128,14 @@ cholesky_hpx(Scheduler &sched, /////////////////////////////////////////////////////////////////////////// // Synchronize - std::vector> result(n_tiles * n_tiles); + std::vector> result(n_tiles * n_tiles); for (std::size_t i = 0; i < n_tiles; i++) { for (std::size_t j = 0; j <= i; j++) { - result[i * n_tiles + j] = tiles[i * n_tiles + j].get_data().get(); + result[i * n_tiles + j] = tiles[i * n_tiles + j].get(); } } - // hpx::get_runtime_distributed().evaluate_active_counters(false, "POST cholesky"); return result; } @@ -218,7 +153,7 @@ gprat_results load_test_data_results(const std::string &filename) } void validate_two_dim_result(const std::vector> &expected, - const std::vector> &actual) + const std::vector> &actual) { if (expected.size() != actual.size()) { @@ -285,7 +220,7 @@ void run(hpx::program_options::variables_map &vm) std::cerr << "We have comparison data!" << std::endl; } - tiled_cholesky_scheduler_distributed scheduler; + scheduler::tiled_cholesky_scheduler_paap12 scheduler; for (std::size_t start = START; start <= END; start = start * STEP) { @@ -295,24 +230,23 @@ void run(hpx::program_options::variables_map &vm) hpx::chrono::high_resolution_timer total_timer; // Compute tile sizes and number of predict tiles - int tile_size = utils::compute_train_tile_size(n_train, n_tiles); - auto result = utils::compute_test_tiles(n_test, n_tiles, tile_size); + int tile_size = compute_train_tile_size(n_train, n_tiles); + auto result = compute_test_tiles(n_test, n_tiles, tile_size); ///////////////////// ///// hyperparams - gprat_hyper::AdamParams hpar = { 0.1, 0.9, 0.999, 1e-8, OPT_ITER }; + AdamParams hpar = { 0.1, 0.9, 0.999, 1e-8, OPT_ITER }; ///////////////////// ////// data loading - gprat::GP_data training_input(train_path, n_train, n_reg); - gprat::GP_data training_output(out_path, n_train, n_reg); - gprat::GP_data test_input(test_path, n_test, n_reg); + GP_data training_input(train_path, n_train, n_reg); + GP_data training_output(out_path, n_train, n_reg); + GP_data test_input(test_path, n_test, n_reg); ///////////////////// ///// GP hpx::chrono::high_resolution_timer init_timer; std::vector trainable = { true, true, true }; - gprat::GP gp( - training_input.data, training_output.data, n_tiles, tile_size, n_reg, { 1.0, 1.0, 0.1 }, trainable); + GP gp(training_input.data, training_output.data, n_tiles, tile_size, n_reg, { 1.0, 1.0, 0.1 }, trainable); const auto init_time = init_timer.elapsed(); // Measure the time taken to execute gp.cholesky(); @@ -338,6 +272,7 @@ void run(hpx::program_options::variables_map &vm) if (test_results) { + std::cerr << "Validating results..." << std::endl; validate_two_dim_result(test_results->choleksy, cholesky); } } @@ -345,6 +280,10 @@ void run(hpx::program_options::variables_map &vm) std::cerr << "DONE!" << std::endl; } +GPRAT_NS_END + +HPX_REGISTER_ACTION(GPRAT_NS::gen_tile_covariance_distributed_action); + int hpx_main(hpx::program_options::variables_map &vm) { std::cerr << "OS Threads: " << hpx::get_os_thread_count() << std::endl; @@ -353,9 +292,10 @@ int hpx_main(hpx::program_options::variables_map &vm) std::cerr << "This locality: " << hpx::find_here() << std::endl; std::cerr << "Remote localities: " << hpx::find_remote_localities().size() << std::endl; + auto numa_domains = hpx::compute::host::numa_domains(); try { - run(vm); + GPRAT_NS::run(vm); } catch (const std::exception &e) { @@ -366,8 +306,8 @@ int hpx_main(hpx::program_options::variables_map &vm) int main(int argc, char *argv[]) { - hpx::register_startup_function(®ister_distributed_tile_counters); - hpx::register_startup_function(®ister_distributed_blas_counters); + hpx::register_startup_function(&GPRAT_NS::register_distributed_tile_counters); + hpx::register_startup_function(&GPRAT_NS::register_distributed_blas_counters); namespace po = hpx::program_options; po::options_description desc("Allowed options"); @@ -378,7 +318,7 @@ int main(int argc, char *argv[]) ("train_x_path", po::value()->default_value("data/data_1024/training_input.txt"), "training data (x)") ("train_y_path", po::value()->default_value("data/data_1024/training_output.txt"), "training data (y)") ("test_path", po::value()->default_value("data/data_1024/test_input.txt"), "test data") - ("test_results_path", po::value(), "test data results to validate results with") + ("test_results_path", po::value()->default_value("data/data_1024/output.json"), "test data results to validate results with") ("timings_csv", po::value()->default_value("timings.csv"), "output timing reports") ("tiles", po::value()->default_value(16), "tiles per dimension") ("regressors", po::value()->default_value(8), "num regressors") diff --git a/examples/distributed/src/scheduling.cpp b/examples/distributed/src/scheduling.cpp new file mode 100644 index 00000000..3917124d --- /dev/null +++ b/examples/distributed/src/scheduling.cpp @@ -0,0 +1,67 @@ +#include "scheduling.hpp" + +#include +#include + +std::atomic tile_transmission_time(0); + +void record_transmission_time(std::int64_t elapsed_ns) +{ + HPX_ASSERT(elapsed_ns >= 0); + tile_transmission_time += elapsed_ns; +} + +std::uint64_t get_transmission_time(bool reset) +{ + return hpx::util::get_and_reset_value(tile_transmission_time, reset); +} + +void register_distributed_tile_counters() +{ + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/hits", + &tile_cache_counters::get_cache_hits, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/misses", + &tile_cache_counters::get_cache_misses, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/evictions", + &tile_cache_counters::get_cache_evictions, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/insertions", + &tile_cache_counters::get_cache_insertions, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/transmission_time", + &get_transmission_time, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); +} + +// The macros below are necessary to generate the code required for exposing +// our partition type remotely. +// +// HPX_REGISTER_COMPONENT() exposes the component creation +// through hpx::new_<>(). +typedef hpx::components::component tile_server_type; +HPX_REGISTER_COMPONENT(tile_server_type, tile_server) + +// HPX_REGISTER_ACTION() exposes the component member function for remote +// invocation. +typedef tile_server::get_data_action get_data_action; +HPX_REGISTER_ACTION(get_data_action) + +typedef tile_server::set_data_action set_data_action; +HPX_REGISTER_ACTION(set_data_action) diff --git a/examples/distributed/src/scheduling.hpp b/examples/distributed/src/scheduling.hpp index 12dcef88..ef8e4ce5 100644 --- a/examples/distributed/src/scheduling.hpp +++ b/examples/distributed/src/scheduling.hpp @@ -1,10 +1,15 @@ #pragma once +#include "gprat/detail/config.hpp" + #include #include #include +#include #include +GPRAT_NS_BEGIN + template struct plain_action_for; @@ -20,35 +25,102 @@ struct plain_action_for; #define GPRAT_DEFINE_PLAIN_ACTION_FOR(local_function) \ std::atomic plain_action_for::elapsed_ns_in_action(0) +struct plain_action_timer +{ + explicit plain_action_timer(std::atomic &total) : + total(total) + { } + + ~plain_action_timer() + { + const auto elapsed = timer.elapsed_nanoseconds(); + HPX_ASSERT(elapsed >= 0); + total += elapsed; + } + + std::atomic &total; + hpx::chrono::high_resolution_timer timer; +}; + +#define GPRAT_TIME_PLAIN_ACTION(local_function) \ + plain_action_timer _action_timer(plain_action_for::elapsed_ns_in_action); + template std::uint64_t get_and_reset_plain_action_elapsed(bool reset) { return hpx::util::get_and_reset_value(plain_action_for::elapsed_ns_in_action, reset); } -// This is a simple tag-type like construct that exists solely, so we automatically pick the right dataflow() overload. -struct schedule_on_locality +struct tiled_scheduler_distributed { - // conversion is intended here, we don't want people to actually spell this type out - // ReSharper disable once CppNonExplicitConvertingConstructor - schedule_on_locality(const hpx::id_type &where) : - where(where) - { } + tiled_scheduler_distributed() : + localities_(hpx::find_all_localities()) + { + // ctor + } - hpx::id_type where; + explicit tiled_scheduler_distributed(std::vector in_localities) : + localities_(std::move(in_localities)) + { + // ctor + } + + std::vector localities_; }; +struct tiled_scheduler_local +{ }; + +namespace detail +{ +// HPX does not auto-collapse future chains in their async(), dataflow(), ... functions. +// This usually works fine, but we require shared_futures most of the time +// and the language will not do two-step conversions for us (future> -> future -> shared_future). +// see: https://github.com/STEllAR-GROUP/hpx/issues/3758 +template +hpx::future collapse(hpx::future> &&fut) +{ + return { std::move(fut) }; +} + +template +hpx::future collapse(hpx::future &&fut) +{ + return std::move(fut); +} + template -decltype(auto) dataflow(const schedule_on_locality &on, Args &&...args) +decltype(auto) +named_dataflow(const tiled_scheduler_distributed &sched, std::size_t on, const char * /*name*/, Args &&...args) { - typename plain_action_for::action_type act; - return hpx::dataflow( + return collapse(hpx::dataflow(hpx::launch::async, + hpx::unwrapping(typename plain_action_for::action_type{}), + sched.localities_[on], + std::forward(args)...)); + /*return hpx::dataflow( [timer = hpx::chrono::high_resolution_timer()](auto &&r) { const auto elapsed = timer.elapsed_nanoseconds(); HPX_ASSERT(elapsed >= 0); plain_action_for::elapsed_ns_in_action += elapsed; - return r.get(); + return collapse(r.get()); }, - hpx::dataflow(act, on.where, args...)); + hpx::dataflow(typename plain_action_for::action_type{}, sched.localities_[on], + std::forward(args)...));*/ +} + +template +decltype(auto) named_dataflow(const tiled_scheduler_local &sched, std::size_t /*on*/, const char *name, Args &&...args) +{ + return hpx::dataflow(hpx::annotated_function(hpx::unwrapping(F), name), std::forward(args)...); +} + +template +decltype(auto) named_async(const tiled_scheduler_local &sched, std::size_t /*on*/, const char *name, Args &&...args) +{ + return hpx::async(hpx::annotated_function(F, name), std::forward(args)...); } + +} // namespace detail + +GPRAT_NS_END diff --git a/test/src/output_correctness.cpp b/test/src/output_correctness.cpp index f924e8f8..89dee4af 100644 --- a/test/src/output_correctness.cpp +++ b/test/src/output_correctness.cpp @@ -1,7 +1,7 @@ -#include "test_data.hpp" #include "gprat/gprat.hpp" #include "gprat/utils.hpp" +#include "test_data.hpp" #include #include From 6018a493d6b09073a78d6caf51dbb0ec9ace6da9 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Thu, 24 Jul 2025 00:09:44 +0200 Subject: [PATCH 38/56] fix(examples): Correct performance counter initialization --- examples/distributed/src/distributed_tile.cpp | 18 ------------------ examples/distributed/src/distributed_tile.hpp | 8 ++++---- examples/distributed/src/main.cpp | 1 + 3 files changed, 5 insertions(+), 22 deletions(-) diff --git a/examples/distributed/src/distributed_tile.cpp b/examples/distributed/src/distributed_tile.cpp index dbc122a5..4d7deba4 100644 --- a/examples/distributed/src/distributed_tile.cpp +++ b/examples/distributed/src/distributed_tile.cpp @@ -39,10 +39,6 @@ void record_transmission_time(std::int64_t elapsed_ns) tile_transmission_time += elapsed_ns; } -void track_tile_data_allocation(std::size_t size) { tile_data_allocations += 1; } - -void track_tile_data_deallocation(std::size_t size) { tile_data_deallocations += 1; } - void track_tile_server_allocation(std::size_t size) { tile_server_allocations += 1; } void track_tile_server_deallocation(std::size_t size) { tile_server_deallocations += 1; } @@ -51,8 +47,6 @@ void track_tile_server_deallocation(std::size_t size) { tile_server_deallocation std::uint64_t get_##name(bool reset) { return hpx::util::get_and_reset_value(name, reset); } GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_transmission_time) -GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_data_allocations) -GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_data_deallocations) GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_server_allocations) GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_server_deallocations) @@ -91,18 +85,6 @@ void register_distributed_tile_counters() "", "", hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_data/num_allocations", - &get_tile_data_allocations, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_data/num_deallocations", - &get_tile_data_deallocations, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); hpx::performance_counters::install_counter_type( "/gprat/tile_server/num_allocations", &get_tile_server_allocations, diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index 74c6a93a..80e15e72 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -20,9 +20,6 @@ GPRAT_NS_BEGIN void register_distributed_tile_counters(); void record_transmission_time(std::int64_t elapsed_ns); -void track_tile_data_allocation(std::size_t size); -void track_tile_data_deallocation(std::size_t size); - void track_tile_server_allocation(std::size_t size); void track_tile_server_deallocation(std::size_t size); @@ -91,7 +88,10 @@ namespace server template struct tile_server : hpx::components::locking_hook>> { - tile_server() = default; + tile_server() + { + track_tile_server_allocation(0); + } explicit tile_server(const mutable_tile_data &data) : data_(data) diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index 52374a80..bcf1cba7 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -306,6 +306,7 @@ int hpx_main(hpx::program_options::variables_map &vm) int main(int argc, char *argv[]) { + hpx::register_startup_function(&GPRAT_NS::register_performance_counters); hpx::register_startup_function(&GPRAT_NS::register_distributed_tile_counters); hpx::register_startup_function(&GPRAT_NS::register_distributed_blas_counters); From f9ef1fbe80f68ee5c9ef95a2102c2f2a90434e1f Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Thu, 24 Jul 2025 04:05:42 +0200 Subject: [PATCH 39/56] chore(examples): Manually unwrap distributed BLAS arguments --- examples/distributed/src/distributed_blas.cpp | 48 +++++++++---------- 1 file changed, 23 insertions(+), 25 deletions(-) diff --git a/examples/distributed/src/distributed_blas.cpp b/examples/distributed/src/distributed_blas.cpp index ce74602b..84665a24 100644 --- a/examples/distributed/src/distributed_blas.cpp +++ b/examples/distributed/src/distributed_blas.cpp @@ -21,12 +21,11 @@ hpx::future> potrf_distributed(const tile_handle &A, { return hpx::dataflow( hpx::launch::async, - hpx::unwrapping( - [A, N](const mutable_tile_data &tile) - { - GPRAT_TIME_PLAIN_ACTION(potrf); - return A.set_async(potrf(tile, N)); - }), + [A, N](hpx::future> &&tile) + { + GPRAT_TIME_PLAIN_ACTION(potrf); + return A.set_async(potrf(tile.get(), N)); + }, A.get_async()); } @@ -40,12 +39,12 @@ hpx::future> trsm_distributed( { return hpx::dataflow( hpx::launch::async, - hpx::unwrapping( - [A, N, M, transpose_L, side_L](const mutable_tile_data &Ld, mutable_tile_data Ad) - { - GPRAT_TIME_PLAIN_ACTION(trsm); - return A.set_async(trsm(Ld, Ad, N, M, transpose_L, side_L)); - }), + [A, N, M, transpose_L, side_L]( + hpx::future> &&Ld, hpx::future> &&Ad) + { + GPRAT_TIME_PLAIN_ACTION(trsm); + return A.set_async(trsm(Ld.get(), Ad.get(), N, M, transpose_L, side_L)); + }, L.get_async(), A.get_async()); } @@ -54,12 +53,11 @@ hpx::future> syrk_distributed(const tile_handle &A, { return hpx::dataflow( hpx::launch::async, - hpx::unwrapping( - [A, N](mutable_tile_data Ad, const mutable_tile_data &Bd) - { - GPRAT_TIME_PLAIN_ACTION(syrk); - return A.set_async(syrk(Ad, Bd, N)); - }), + [A, N](hpx::future> &&Ad, hpx::future> &&Bd) + { + GPRAT_TIME_PLAIN_ACTION(syrk); + return A.set_async(syrk(Ad.get(), Bd.get(), N)); + }, A.get_async(), B.get_async()); } @@ -76,13 +74,13 @@ hpx::future> gemm_distributed( { return hpx::dataflow( hpx::launch::async, - hpx::unwrapping( - [C, N, M, K, transpose_A, transpose_B]( - const mutable_tile_data &Ad, const mutable_tile_data &Bd, mutable_tile_data Cd) - { - GPRAT_TIME_PLAIN_ACTION(gemm); - return C.set_async(gemm(Ad, Bd, Cd, N, M, K, transpose_A, transpose_B)); - }), + [C, N, M, K, transpose_A, transpose_B](hpx::future> &&Ad, + hpx::future> &&Bd, + hpx::future> &&Cd) + { + GPRAT_TIME_PLAIN_ACTION(gemm); + return C.set_async(gemm(Ad.get(), Bd.get(), Cd.get(), N, M, K, transpose_A, transpose_B)); + }, A.get_async(), B.get_async(), C.get_async()); From 2c69dae3348e8f3de77523fd2072c95b3bed6420 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Thu, 24 Jul 2025 04:06:57 +0200 Subject: [PATCH 40/56] fix(examples): Ensure we correctly copy training data --- examples/distributed/src/main.cpp | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index bcf1cba7..7794c7c7 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -31,7 +31,7 @@ hpx::future> gen_tile_covariance_distributed( std::size_t N, std::size_t n_regressors, const SEKParams &sek_params, - std::span input) + const std::vector& input) { return tile.set_async(cpu::gen_tile_covariance(row, col, N, n_regressors, sek_params, input)); } @@ -96,7 +96,7 @@ void right_looking_cholesky_tiled(Scheduler &sched, Tiles &ft_tiles, std::size_t template std::vector> cholesky_hpx(Scheduler &sched, - std::span training_input, + const std::vector &training_input, const SEKParams &sek_params, std::size_t n_tiles, std::size_t n_tile_size, From 558baf1a19159fe66fa7f58205e293eeb8d31a05 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Thu, 24 Jul 2025 05:19:25 +0200 Subject: [PATCH 41/56] fix(examples): Use modules instead of startup functions The latter seems to have race conditions. --- examples/distributed/src/main.cpp | 24 +++++++++++++++++++----- 1 file changed, 19 insertions(+), 5 deletions(-) diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index 7794c7c7..35129bf3 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -31,7 +31,7 @@ hpx::future> gen_tile_covariance_distributed( std::size_t N, std::size_t n_regressors, const SEKParams &sek_params, - const std::vector& input) + const std::vector &input) { return tile.set_async(cpu::gen_tile_covariance(row, col, N, n_regressors, sek_params, input)); } @@ -280,10 +280,28 @@ void run(hpx::program_options::variables_map &vm) std::cerr << "DONE!" << std::endl; } +// startup function for PAPI counter component +void startup() +{ + register_performance_counters(); + register_distributed_tile_counters(); + register_distributed_blas_counters(); +} + +bool check_startup(hpx::startup_function_type &startup_func, bool &pre_startup) +{ + // perform full module startup (counters will be used) + startup_func = startup; + pre_startup = true; + return true; +} + GPRAT_NS_END HPX_REGISTER_ACTION(GPRAT_NS::gen_tile_covariance_distributed_action); +HPX_REGISTER_STARTUP_MODULE_DYNAMIC(GPRAT_NS::check_startup) + int hpx_main(hpx::program_options::variables_map &vm) { std::cerr << "OS Threads: " << hpx::get_os_thread_count() << std::endl; @@ -306,10 +324,6 @@ int hpx_main(hpx::program_options::variables_map &vm) int main(int argc, char *argv[]) { - hpx::register_startup_function(&GPRAT_NS::register_performance_counters); - hpx::register_startup_function(&GPRAT_NS::register_distributed_tile_counters); - hpx::register_startup_function(&GPRAT_NS::register_distributed_blas_counters); - namespace po = hpx::program_options; po::options_description desc("Allowed options"); From dda28da8eca50c7041aa1978aeb617ddd46d145c Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Thu, 24 Jul 2025 05:19:52 +0200 Subject: [PATCH 42/56] fix(examples): Don't depend on the accessor staying alive --- examples/distributed/src/distributed_tile.hpp | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index 80e15e72..3a74c58f 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -257,16 +257,17 @@ class tiled_dataset_accessor hpx::future> set_tile_data(std::size_t tile_index, std::size_t generation, const mutable_tile_data &data) const { + auto self = base_type::get(); if (tiles_[tile_index].local_data) { tiles_[tile_index].local_data->set_data(data); - return hpx::make_ready_future(tile_handle{ base_type::get(), tile_index, generation + 1 }); + return hpx::make_ready_future(tile_handle{ std::move(self), tile_index, generation + 1 }); } typename server::tile_server::set_data_action act; return hpx::async(act, tiles_[tile_index].tile, data) - .then([this, tile_index, generation](const hpx::future &) - { return tile_handle{ base_type::get(), tile_index, generation + 1 }; }); + .then([self = std::move(self), tile_index, generation](const hpx::future &) + { return tile_handle{ std::move(self), tile_index, generation + 1 }; }); } // TRANSITION From bed699a427ba3ac1fb296d25cf7b040aaa8f2bae Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Thu, 24 Jul 2025 05:28:34 +0200 Subject: [PATCH 43/56] fix(examples): Use static startup modules --- examples/distributed/src/main.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index 35129bf3..0bcacb8a 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -300,7 +300,7 @@ GPRAT_NS_END HPX_REGISTER_ACTION(GPRAT_NS::gen_tile_covariance_distributed_action); -HPX_REGISTER_STARTUP_MODULE_DYNAMIC(GPRAT_NS::check_startup) +HPX_REGISTER_STARTUP_MODULE(GPRAT_NS::check_startup) int hpx_main(hpx::program_options::variables_map &vm) { From bc4130b6ff03f5b07102c09315df8c6cce4596cb Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Fri, 25 Jul 2025 00:29:07 +0200 Subject: [PATCH 44/56] chore(examples): Add more perf counters --- examples/distributed/src/main.cpp | 22 +++++++++++++++++----- 1 file changed, 17 insertions(+), 5 deletions(-) diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index 0bcacb8a..356f2834 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -31,16 +31,25 @@ hpx::future> gen_tile_covariance_distributed( std::size_t N, std::size_t n_regressors, const SEKParams &sek_params, - const std::vector &input) -{ - return tile.set_async(cpu::gen_tile_covariance(row, col, N, n_regressors, sek_params, input)); -} - + const std::vector &input); HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_covariance_distributed); GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_covariance, gen_tile_covariance_distributed_action, "gen_tile_covariance"); +hpx::future> gen_tile_covariance_distributed( + tile_handle tile, + std::size_t row, + std::size_t col, + std::size_t N, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &input) +{ + GPRAT_TIME_PLAIN_ACTION(cpu::gen_tile_covariance); + return tile.set_async(cpu::gen_tile_covariance(row, col, N, n_regressors, sek_params, input)); +} + template void right_looking_cholesky_tiled(Scheduler &sched, Tiles &ft_tiles, std::size_t N, std::size_t n_tiles) { @@ -304,6 +313,7 @@ HPX_REGISTER_STARTUP_MODULE(GPRAT_NS::check_startup) int hpx_main(hpx::program_options::variables_map &vm) { + hpx::get_runtime().get_config().dump(0, std::cerr); std::cerr << "OS Threads: " << hpx::get_os_thread_count() << std::endl; std::cerr << "All localities: " << hpx::get_num_localities().get() << std::endl; std::cerr << "Root locality: " << hpx::find_root_locality() << std::endl; @@ -311,6 +321,8 @@ int hpx_main(hpx::program_options::variables_map &vm) std::cerr << "Remote localities: " << hpx::find_remote_localities().size() << std::endl; auto numa_domains = hpx::compute::host::numa_domains(); + std::cerr << "Local NUMA domains: " << numa_domains.size() << std::endl; + try { GPRAT_NS::run(vm); From 61967cf7416477adfefff65a1125b11bda373de8 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Fri, 25 Jul 2025 00:36:36 +0200 Subject: [PATCH 45/56] fix(examples): Add missing definition --- examples/distributed/src/main.cpp | 1 + 1 file changed, 1 insertion(+) diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index 356f2834..f6b8fecf 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -36,6 +36,7 @@ HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_covariance_distributed); GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_covariance, gen_tile_covariance_distributed_action, "gen_tile_covariance"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_covariance); hpx::future> gen_tile_covariance_distributed( tile_handle tile, From e50978bd02e5fc8117c3f666bff83396b0fe9adf Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Fri, 25 Jul 2025 03:46:58 +0200 Subject: [PATCH 46/56] feat(examples): Re-add cache for new implementation --- examples/distributed/src/distributed_tile.cpp | 28 +----- examples/distributed/src/distributed_tile.hpp | 98 ++++++++++++++++--- examples/distributed/src/main.cpp | 20 +++- examples/distributed/src/scheduling.hpp | 3 +- 4 files changed, 110 insertions(+), 39 deletions(-) diff --git a/examples/distributed/src/distributed_tile.cpp b/examples/distributed/src/distributed_tile.cpp index 4d7deba4..a5864ad0 100644 --- a/examples/distributed/src/distributed_tile.cpp +++ b/examples/distributed/src/distributed_tile.cpp @@ -7,14 +7,13 @@ HPX_DISTRIBUTED_METADATA(GPRAT_NS::server::tiled_dataset_config_data, gprat_serv GPRAT_NS_BEGIN -/* struct tile_cache_counters { // XXX: you can do this with templates, but it's quite a bit more complicated #define GPRAT_MAKE_STATISTICS_ACCESSOR(name) \ static std::uint64_t get_cache_##name(bool reset) \ { \ - auto &cache = get_tile_cache(); \ + auto &cache = get_tile_cache(); \ std::lock_guard lock(cache.mutex_); \ return cache.cache_.get_statistics().name(reset); \ } @@ -26,7 +25,7 @@ struct tile_cache_counters #undef GPRAT_MAKE_STATISTICS_ACCESSOR }; -*/ + std::atomic tile_transmission_time(0); std::atomic tile_data_allocations(0); std::atomic tile_data_deallocations(0); @@ -53,7 +52,7 @@ GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_server_deallocations) #undef GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR void register_distributed_tile_counters() -{ /* +{ hpx::performance_counters::install_counter_type( "/gprat/tile_cache/hits", &tile_cache_counters::get_cache_hits, @@ -78,13 +77,14 @@ void register_distributed_tile_counters() "", "", hpx::performance_counters::counter_type::monotonically_increasing); - */ hpx::performance_counters::install_counter_type( "/gprat/tile_cache/transmission_time", &get_tile_transmission_time, "", "", hpx::performance_counters::counter_type::monotonically_increasing); + + hpx::performance_counters::install_counter_type( "/gprat/tile_server/num_allocations", &get_tile_server_allocations, @@ -99,22 +99,4 @@ void register_distributed_tile_counters() hpx::performance_counters::counter_type::monotonically_increasing); } -/* -tile_cache::tile_cache() : - cache_(16) -{ } - -bool tile_cache::try_get(const hpx::naming::gid_type &key, tile_data &cached_data) -{ - std::lock_guard g(mutex_); - hpx::naming::gid_type unused; - return cache_.get_entry(key, unused, cached_data); -} - -void tile_cache::insert(const hpx::naming::gid_type &key, const tile_data &data) -{ - std::lock_guard g(mutex_); - cache_.insert(key, data); -}*/ - GPRAT_NS_END diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index 3a74c58f..1c5743ec 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -80,6 +80,59 @@ HPX_DISTRIBUTED_METADATA_DECLARATION(GPRAT_NS::server::tiled_dataset_config_data GPRAT_NS_BEGIN +template +class tile_cache +{ + friend struct tile_cache_counters; + + public: + tile_cache() : + cache_(16) + { } + + bool try_get(const hpx::naming::gid_type &key, std::size_t generation, mutable_tile_data &cached_data) + { + std::lock_guard g(mutex_); + hpx::naming::gid_type unused; + entry e; + if (cache_.get_entry(key, unused, e)) + { + if (e.generation == generation) + { + cached_data = e.data; + return true; + } + // Erase the obsolete entry + cache_.erase([&](const auto &p) { return p.first == key; }); + } + return false; + } + + void insert(const hpx::naming::gid_type &key, std::size_t generation, const mutable_tile_data &data) + { + std::lock_guard g(mutex_); + cache_.insert(key, entry{ data, generation }); + } + + private: + struct entry + { + mutable_tile_data data; + std::size_t generation = 0; + }; + + hpx::mutex mutex_; + hpx::util::cache::lru_cache + cache_; +}; + +template +tile_cache &get_tile_cache() +{ + static tile_cache cache; + return cache; +} + namespace server { // This is the server side representation of the data. We expose this as a HPX @@ -88,10 +141,7 @@ namespace server template struct tile_server : hpx::components::locking_hook>> { - tile_server() - { - track_tile_server_allocation(0); - } + tile_server() { track_tile_server_allocation(0); } explicit tile_server(const mutable_tile_data &data) : data_(data) @@ -243,29 +293,55 @@ class tiled_dataset_accessor { return assign_existing(id, f.get()); }); } - hpx::future> get_tile_data(std::size_t tile_index, std::size_t /*generation*/) const + hpx::future> get_tile_data(std::size_t tile_index, std::size_t generation) const { - if (tiles_[tile_index].local_data) + const auto &target_tile = tiles_[tile_index]; + + // Best is always to rely on local data + if (target_tile.local_data) { - return hpx::make_ready_future(tiles_[tile_index].local_data->get_data()); + return hpx::make_ready_future(target_tile.local_data->get_data()); + } + + // Next, try the tile cache - maybe we have current data + { + mutable_tile_data cached_data; + if (get_tile_cache().try_get(target_tile.tile.get_gid(), generation, cached_data)) + { + return hpx::make_ready_future(cached_data); + } } typename server::tile_server::get_data_action act; - return hpx::async(act, tiles_[tile_index].tile); + return hpx::async(act, target_tile.tile) + .then( + [generation, gid = target_tile.tile.get_gid(), timer = hpx::chrono::high_resolution_timer()]( + hpx::future> &&f) + { + record_transmission_time(timer.elapsed_nanoseconds()); + auto data = f.get(); + get_tile_cache().insert(gid, generation, data); + return data; + }); } hpx::future> set_tile_data(std::size_t tile_index, std::size_t generation, const mutable_tile_data &data) const { + const auto &target_tile = tiles_[tile_index]; + auto self = base_type::get(); - if (tiles_[tile_index].local_data) + if (target_tile.local_data) { - tiles_[tile_index].local_data->set_data(data); + target_tile.local_data->set_data(data); return hpx::make_ready_future(tile_handle{ std::move(self), tile_index, generation + 1 }); } + // We'd lose this tile after writing it, best to put it in the cache for now + get_tile_cache().insert(target_tile.tile.get_gid(), generation, data); + typename server::tile_server::set_data_action act; - return hpx::async(act, tiles_[tile_index].tile, data) + return hpx::async(act, target_tile.tile, data) .then([self = std::move(self), tile_index, generation](const hpx::future &) { return tile_handle{ std::move(self), tile_index, generation + 1 }; }); } diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index f6b8fecf..68910125 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -290,12 +290,19 @@ void run(hpx::program_options::variables_map &vm) std::cerr << "DONE!" << std::endl; } -// startup function for PAPI counter component void startup() { - register_performance_counters(); - register_distributed_tile_counters(); - register_distributed_blas_counters(); + std::cerr << "startup() called" << std::endl; + + static struct once_dummy_struct + { + once_dummy_struct() + { + register_performance_counters(); + register_distributed_tile_counters(); + register_distributed_blas_counters(); + } + } once_dummy; } bool check_startup(hpx::startup_function_type &startup_func, bool &pre_startup) @@ -323,6 +330,11 @@ int hpx_main(hpx::program_options::variables_map &vm) auto numa_domains = hpx::compute::host::numa_domains(); std::cerr << "Local NUMA domains: " << numa_domains.size() << std::endl; + for (const auto &domain : numa_domains) + { + const auto &num_pus = domain.num_pus(); + std::cerr << " Domain: " << num_pus.first << " " << num_pus.second << std::endl; + } try { diff --git a/examples/distributed/src/scheduling.hpp b/examples/distributed/src/scheduling.hpp index ef8e4ce5..919a251b 100644 --- a/examples/distributed/src/scheduling.hpp +++ b/examples/distributed/src/scheduling.hpp @@ -35,7 +35,8 @@ struct plain_action_timer { const auto elapsed = timer.elapsed_nanoseconds(); HPX_ASSERT(elapsed >= 0); - total += elapsed; + if (elapsed > 0) + total += static_cast(elapsed); } std::atomic &total; From 714cea59f7de26ad555dc895be9c5f55ec5b1765 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Fri, 25 Jul 2025 23:26:48 +0200 Subject: [PATCH 47/56] feat(examples): Add cyclic cholesky scheduler --- .../distributed/src/distributed_cholesky.hpp | 104 ++++++++++++++---- examples/distributed/src/main.cpp | 33 +++--- 2 files changed, 98 insertions(+), 39 deletions(-) diff --git a/examples/distributed/src/distributed_cholesky.hpp b/examples/distributed/src/distributed_cholesky.hpp index 2f34e4eb..96435c17 100644 --- a/examples/distributed/src/distributed_cholesky.hpp +++ b/examples/distributed/src/distributed_cholesky.hpp @@ -24,10 +24,6 @@ constexpr std::size_t cholesky_TRSM(...) { return 0; } constexpr std::size_t cholesky_GEMM(...) { return 0; } -constexpr std::size_t cholesky_TRSV(...) { return 0; } - -constexpr std::size_t cholesky_GEMV(...) { return 0; } - namespace scheduler { @@ -39,21 +35,21 @@ struct tiled_cholesky_scheduler_paap12 : tiled_scheduler_distributed }; template -tiled_dataset make_cholesky_dataset(const tiled_cholesky_scheduler_paap12 &policy, std::size_t num_tiles) +tiled_dataset make_cholesky_dataset(const tiled_cholesky_scheduler_paap12 &sched, std::size_t num_tiles) { std::vector> targets; - targets.reserve(policy.num_localities); + targets.reserve(sched.num_localities); - for (std::size_t i = 0; i < policy.num_localities; ++i) + for (std::size_t i = 0; i < sched.num_localities; ++i) { - targets.emplace_back(policy.localities_[i], 0); + targets.emplace_back(sched.localities_[i], 0); } for (std::size_t row = 0; row < num_tiles; row++) { for (std::size_t col = 0; col < num_tiles; col++) { - const auto l = (row + col) % policy.num_localities; + const auto l = (row + col) % sched.num_localities; ++targets[l].second; } } @@ -61,40 +57,102 @@ tiled_dataset make_cholesky_dataset(const tiled_cholesky_scheduler_paap12 &po return tiled_dataset_accessor{ targets, num_tiles * num_tiles }.to_dataset(); } -constexpr std::size_t cholesky_tile(const tiled_cholesky_scheduler_paap12 &policy, std::size_t row, std::size_t col) +constexpr std::size_t +cholesky_tile(const tiled_cholesky_scheduler_paap12 &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) { - return (row + col) % policy.num_localities; + return (row + col) % sched.num_localities; } -constexpr std::size_t cholesky_POTRF(const tiled_cholesky_scheduler_paap12 &policy, std::size_t k) +constexpr std::size_t +cholesky_POTRF(const tiled_cholesky_scheduler_paap12 &sched, std::size_t /*n_tiles*/, std::size_t k) { - return (2 * k) % policy.num_localities; + return (2 * k) % sched.num_localities; } -constexpr std::size_t cholesky_SYRK(const tiled_cholesky_scheduler_paap12 &policy, std::size_t m) +constexpr std::size_t +cholesky_SYRK(const tiled_cholesky_scheduler_paap12 &sched, std::size_t /*n_tiles*/, std::size_t m) { - return (2 * m) % policy.num_localities; + return (2 * m) % sched.num_localities; } -constexpr std::size_t cholesky_TRSM(const tiled_cholesky_scheduler_paap12 &policy, std::size_t k, std::size_t m) +constexpr std::size_t +cholesky_TRSM(const tiled_cholesky_scheduler_paap12 &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) { - return (k + m) % policy.num_localities; + return (k + m) % sched.num_localities; +} + +constexpr std::size_t cholesky_GEMM(const tiled_cholesky_scheduler_paap12 &sched, + std::size_t /*n_tiles*/, + std::size_t /*k*/, + std::size_t m, + std::size_t n) +{ + return (m + n) % sched.num_localities; +} + +// ========================== + +struct tiled_cholesky_scheduler_cyclic : tiled_scheduler_distributed +{ + using tiled_scheduler_distributed::tiled_scheduler_distributed; + + std::size_t num_localities = localities_.size(); +}; + +template +tiled_dataset make_cholesky_dataset(const tiled_cholesky_scheduler_cyclic &sched, std::size_t num_tiles) +{ + std::vector> targets; + targets.reserve(sched.num_localities); + + for (std::size_t i = 0; i < sched.num_localities; ++i) + { + targets.emplace_back(sched.localities_[i], 0); + } + + for (std::size_t row = 0; row < num_tiles; row++) + { + for (std::size_t col = 0; col < num_tiles; col++) + { + const auto l = (row * num_tiles + col) % sched.num_localities; + ++targets[l].second; + } + } + + return tiled_dataset_accessor{ targets, num_tiles * num_tiles }.to_dataset(); +} + +constexpr std::size_t +cholesky_tile(const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +{ + return (row * n_tiles + col) % sched.num_localities; } constexpr std::size_t -cholesky_GEMM(const tiled_cholesky_scheduler_paap12 &policy, std::size_t /*k*/, std::size_t m, std::size_t n) +cholesky_POTRF(const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) { - return (m + n) % policy.num_localities; + return (k * n_tiles + k) % sched.num_localities; } -constexpr std::size_t cholesky_TRSV(const tiled_cholesky_scheduler_paap12 &policy, std::size_t k) +constexpr std::size_t +cholesky_SYRK(const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t m) +{ + return (m * n_tiles + m) % sched.num_localities; +} + +constexpr std::size_t +cholesky_TRSM(const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k, std::size_t m) { - return k % policy.num_localities; + return (m * n_tiles + k) % sched.num_localities; } -constexpr std::size_t cholesky_GEMV(const tiled_cholesky_scheduler_paap12 &policy, std::size_t k) +constexpr std::size_t cholesky_GEMM(const tiled_cholesky_scheduler_cyclic &sched, + std::size_t n_tiles, + std::size_t /*k*/, + std::size_t m, + std::size_t n) { - return k % policy.num_localities; + return (m * n_tiles + n) % sched.num_localities; } } // namespace scheduler diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index 68910125..8097e351 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -16,6 +16,7 @@ // Better than having the whole project depend on compiled Boost.Json! #include "gprat/gprat.hpp" +#include "gprat/performance_counters.hpp" #include "gprat/utils.hpp" #include @@ -52,22 +53,22 @@ hpx::future> gen_tile_covariance_distributed( } template -void right_looking_cholesky_tiled(Scheduler &sched, Tiles &ft_tiles, std::size_t N, std::size_t n_tiles) +void right_looking_cholesky_tiled(Scheduler &sched, Tiles &tiles, std::size_t N, std::size_t n_tiles) { for (std::size_t k = 0; k < n_tiles; k++) { // POTRF: Compute Cholesky factor L - ft_tiles[k * n_tiles + k] = detail::named_dataflow( - sched, cholesky_POTRF(sched, k), "cholesky_tiled", ft_tiles[k * n_tiles + k], N); + tiles[k * n_tiles + k] = detail::named_dataflow( + sched, cholesky_POTRF(sched, n_tiles, k), "cholesky_tiled", tiles[k * n_tiles + k], N); for (std::size_t m = k + 1; m < n_tiles; m++) { // TRSM: Solve X * L^T = A - ft_tiles[m * n_tiles + k] = detail::named_dataflow( + tiles[m * n_tiles + k] = detail::named_dataflow( sched, cholesky_TRSM(sched, k, m), "cholesky_tiled", - ft_tiles[k * n_tiles + k], - ft_tiles[m * n_tiles + k], + tiles[k * n_tiles + k], + tiles[m * n_tiles + k], N, N, Blas_trans, @@ -76,23 +77,23 @@ void right_looking_cholesky_tiled(Scheduler &sched, Tiles &ft_tiles, std::size_t for (std::size_t m = k + 1; m < n_tiles; m++) { // SYRK: A = A - B * B^T - ft_tiles[m * n_tiles + m] = detail::named_dataflow( + tiles[m * n_tiles + m] = detail::named_dataflow( sched, - cholesky_SYRK(sched, m), + cholesky_SYRK(sched, n_tiles, m), "cholesky_tiled", - ft_tiles[m * n_tiles + m], - ft_tiles[m * n_tiles + k], + tiles[m * n_tiles + m], + tiles[m * n_tiles + k], N); for (std::size_t n = k + 1; n < m; n++) { // GEMM: C = C - A * B^T - ft_tiles[m * n_tiles + n] = detail::named_dataflow( + tiles[m * n_tiles + n] = detail::named_dataflow( sched, - cholesky_GEMM(sched, k, m, n), + cholesky_GEMM(sched, n_tiles, k, m, n), "cholesky_tiled", - ft_tiles[m * n_tiles + k], - ft_tiles[n * n_tiles + k], - ft_tiles[m * n_tiles + n], + tiles[m * n_tiles + k], + tiles[n * n_tiles + k], + tiles[m * n_tiles + n], N, N, N, @@ -120,7 +121,7 @@ cholesky_hpx(Scheduler &sched, { tiles[row * n_tiles + col] = detail::named_dataflow( sched, - cholesky_tile(sched, row, col), + cholesky_tile(sched, n_tiles, row, col), "cholesky init", tiles[row * n_tiles + col], row, From fb0eeb511f61926b3ec5a2e474a4fa1620f96f85 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sat, 26 Jul 2025 01:09:45 +0200 Subject: [PATCH 48/56] refactor!(examples): Introduce per-locality manager components --- .../distributed/src/distributed_cholesky.hpp | 18 +- examples/distributed/src/distributed_tile.cpp | 77 +-- examples/distributed/src/distributed_tile.hpp | 518 ++++++++---------- examples/distributed/src/main.cpp | 2 +- 4 files changed, 266 insertions(+), 349 deletions(-) diff --git a/examples/distributed/src/distributed_cholesky.hpp b/examples/distributed/src/distributed_cholesky.hpp index 96435c17..3f4d795a 100644 --- a/examples/distributed/src/distributed_cholesky.hpp +++ b/examples/distributed/src/distributed_cholesky.hpp @@ -2,7 +2,6 @@ #include "distributed_tile.hpp" #include "scheduling.hpp" -#include GPRAT_NS_BEGIN @@ -54,7 +53,7 @@ tiled_dataset make_cholesky_dataset(const tiled_cholesky_scheduler_paap12 &sc } } - return tiled_dataset_accessor{ targets, num_tiles * num_tiles }.to_dataset(); + return create_tiled_dataset(targets, num_tiles * num_tiles); } constexpr std::size_t @@ -119,7 +118,7 @@ tiled_dataset make_cholesky_dataset(const tiled_cholesky_scheduler_cyclic &sc } } - return tiled_dataset_accessor{ targets, num_tiles * num_tiles }.to_dataset(); + return create_tiled_dataset(targets, num_tiles * num_tiles); } constexpr std::size_t @@ -128,14 +127,12 @@ cholesky_tile(const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, return (row * n_tiles + col) % sched.num_localities; } -constexpr std::size_t -cholesky_POTRF(const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +constexpr std::size_t cholesky_POTRF(const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) { return (k * n_tiles + k) % sched.num_localities; } -constexpr std::size_t -cholesky_SYRK(const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t m) +constexpr std::size_t cholesky_SYRK(const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t m) { return (m * n_tiles + m) % sched.num_localities; } @@ -146,11 +143,8 @@ cholesky_TRSM(const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, return (m * n_tiles + k) % sched.num_localities; } -constexpr std::size_t cholesky_GEMM(const tiled_cholesky_scheduler_cyclic &sched, - std::size_t n_tiles, - std::size_t /*k*/, - std::size_t m, - std::size_t n) +constexpr std::size_t cholesky_GEMM( + const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t /*k*/, std::size_t m, std::size_t n) { return (m * n_tiles + n) % sched.num_localities; } diff --git a/examples/distributed/src/distributed_tile.cpp b/examples/distributed/src/distributed_tile.cpp index a5864ad0..f4acfd2b 100644 --- a/examples/distributed/src/distributed_tile.cpp +++ b/examples/distributed/src/distributed_tile.cpp @@ -3,20 +3,23 @@ #include #include -HPX_DISTRIBUTED_METADATA(GPRAT_NS::server::tiled_dataset_config_data, gprat_server_tiled_dataset_config_data) - GPRAT_NS_BEGIN +namespace detail +{ +hpx::util::cache::statistics::local_full_statistics &get_global_statistics() +{ + static hpx::util::cache::statistics::local_full_statistics stats; + return stats; +} + +} // namespace detail + struct tile_cache_counters { // XXX: you can do this with templates, but it's quite a bit more complicated #define GPRAT_MAKE_STATISTICS_ACCESSOR(name) \ - static std::uint64_t get_cache_##name(bool reset) \ - { \ - auto &cache = get_tile_cache(); \ - std::lock_guard lock(cache.mutex_); \ - return cache.cache_.get_statistics().name(reset); \ - } + static std::uint64_t get_cache_##name(bool reset) { return detail::get_global_statistics().name(reset); } GPRAT_MAKE_STATISTICS_ACCESSOR(hits); GPRAT_MAKE_STATISTICS_ACCESSOR(misses); @@ -35,12 +38,15 @@ std::atomic tile_server_deallocations(0); void record_transmission_time(std::int64_t elapsed_ns) { HPX_ASSERT(elapsed_ns >= 0); - tile_transmission_time += elapsed_ns; + if (elapsed_ns > 0) + { + tile_transmission_time += static_cast(elapsed_ns); + } } -void track_tile_server_allocation(std::size_t size) { tile_server_allocations += 1; } +void track_tile_server_allocation(std::size_t /*size*/) { tile_server_allocations += 1; } -void track_tile_server_deallocation(std::size_t size) { tile_server_deallocations += 1; } +void track_tile_server_deallocation(std::size_t /*size*/) { tile_server_deallocations += 1; } #define GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(name) \ std::uint64_t get_##name(bool reset) { return hpx::util::get_and_reset_value(name, reset); } @@ -53,30 +59,30 @@ GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_server_deallocations) void register_distributed_tile_counters() { - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/hits", - &tile_cache_counters::get_cache_hits, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/misses", - &tile_cache_counters::get_cache_misses, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/evictions", - &tile_cache_counters::get_cache_evictions, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/insertions", - &tile_cache_counters::get_cache_insertions, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/hits", + &tile_cache_counters::get_cache_hits, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/misses", + &tile_cache_counters::get_cache_misses, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/evictions", + &tile_cache_counters::get_cache_evictions, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( + "/gprat/tile_cache/insertions", + &tile_cache_counters::get_cache_insertions, + "", + "", + hpx::performance_counters::counter_type::monotonically_increasing); hpx::performance_counters::install_counter_type( "/gprat/tile_cache/transmission_time", &get_tile_transmission_time, @@ -84,7 +90,6 @@ void register_distributed_tile_counters() "", hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( "/gprat/tile_server/num_allocations", &get_tile_server_allocations, diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index 1c5743ec..5f2446f5 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -23,62 +23,30 @@ void record_transmission_time(std::int64_t elapsed_ns); void track_tile_server_allocation(std::size_t size); void track_tile_server_deallocation(std::size_t size); -namespace server -{ -struct tiled_dataset_config_data +namespace detail { - struct tile_entry - { - tile_entry() : - locality_id(hpx::naming::invalid_locality_id), - generation(0) - { } +hpx::util::cache::statistics::local_full_statistics &get_global_statistics(); - tile_entry(hpx::id_type tile, std::uint32_t locality_id, std::uint64_t generation) : - tile(std::move(tile)), - locality_id(locality_id), - generation(generation) - { } - - hpx::id_type tile; - std::uint32_t locality_id; - std::uint64_t generation; - - private: - friend class hpx::serialization::access; +/////////////////////////////////////////////////////////////////////////// +class global_full_statistics +{ + public: + using update_on_exit = hpx::util::cache::statistics::local_full_statistics::update_on_exit; - template - void serialize(Archive &ar, unsigned) - { - ar & tile & locality_id & generation; - } - }; + // ReSharper disable once CppNonExplicitConversionOperator + operator hpx::util::cache::statistics::local_full_statistics &() const { return get_global_statistics(); } - tiled_dataset_config_data() = default; + void got_hit() noexcept { get_global_statistics().got_hit(); } - tiled_dataset_config_data(std::vector &&tiles) : - tiles(std::move(tiles)) - { } + void got_miss() noexcept { get_global_statistics().got_miss(); } - std::vector tiles; + void got_insertion() noexcept { get_global_statistics().got_insertion(); } - private: - friend class hpx::serialization::access; + void got_eviction() noexcept { get_global_statistics().got_eviction(); } - template - void serialize(Archive &ar, unsigned) - { - ar & tiles; - } + void clear() noexcept { get_global_statistics().clear(); } }; -} // namespace server - -GPRAT_NS_END - -HPX_DISTRIBUTED_METADATA_DECLARATION(GPRAT_NS::server::tiled_dataset_config_data, - gprat_server_tiled_dataset_config_data) - -GPRAT_NS_BEGIN +} // namespace detail template class tile_cache @@ -114,6 +82,8 @@ class tile_cache cache_.insert(key, entry{ data, generation }); } + void clear() { cache_.clear(); } + private: struct entry { @@ -122,34 +92,30 @@ class tile_cache }; hpx::mutex mutex_; - hpx::util::cache::lru_cache - cache_; + hpx::util::cache::lru_cache cache_; }; -template -tile_cache &get_tile_cache() -{ - static tile_cache cache; - return cache; -} - namespace server { -// This is the server side representation of the data. We expose this as a HPX -// component which allows for it to be created and accessed remotely through -// a global address (hpx::id_type). + +/** + * Server component owning a single tile's data. + * + * @tparam T Element type of the tile. Usually some numeric type like double or float. This class currently only + * requires T to be serializable by HPX. + */ template -struct tile_server : hpx::components::locking_hook>> +struct tile_holder : hpx::components::locking_hook>> { - tile_server() { track_tile_server_allocation(0); } + tile_holder() { track_tile_server_allocation(0); } - explicit tile_server(const mutable_tile_data &data) : + explicit tile_holder(const mutable_tile_data &data) : data_(data) { track_tile_server_allocation(data.size()); } - ~tile_server() { track_tile_server_deallocation(data_.size()); } + ~tile_holder() { track_tile_server_deallocation(data_.size()); } [[nodiscard]] mutable_tile_data get_data() const { return data_; } @@ -157,145 +123,91 @@ struct tile_server : hpx::components::locking_hook data_; }; -} // namespace server - -#define GPRAT_REGISTER_TILED_DATASET_DECLARATION_IMPL(type, name) \ - HPX_REGISTER_ACTION_DECLARATION(type::get_data_action, HPX_PP_CAT(_tiled_dataset_get_data_action_, name)) \ - HPX_REGISTER_ACTION_DECLARATION(type::set_data_action, HPX_PP_CAT(_tiled_dataset_set_data_action_, name)) \ - /**/ - -#define GPRAT_REGISTER_TILED_DATASET_DECLARATION(type, name) \ - typedef ::GPRAT_NS::server::tile_server HPX_PP_CAT(_tiled_dataset_server_, HPX_PP_CAT(type, name)); \ - GPRAT_REGISTER_TILED_DATASET_DECLARATION_IMPL(HPX_PP_CAT(_tiled_dataset_server_, HPX_PP_CAT(type, name)), name) - -#define GPRAT_REGISTER_TILED_DATASET_IMPL(type, name) \ - HPX_REGISTER_ACTION(type::get_data_action, HPX_PP_CAT(_tiled_dataset_get_data_action_, name)) \ - HPX_REGISTER_ACTION(type::set_data_action, HPX_PP_CAT(_tiled_dataset_set_data_action_, name)) \ - typedef ::hpx::components::component HPX_PP_CAT(_tiled_dataset_server_component_, name); \ - HPX_REGISTER_COMPONENT(HPX_PP_CAT(_tiled_dataset_server_component_, name)) \ - /**/ - -#define GPRAT_REGISTER_TILED_DATASET(type, name) \ - typedef ::GPRAT_NS::server::tile_server HPX_PP_CAT(_tiled_dataset_server_, HPX_PP_CAT(type, name)); \ - GPRAT_REGISTER_TILED_DATASET_IMPL(HPX_PP_CAT(_tiled_dataset_server_, HPX_PP_CAT(type, name)), name) - template -class tiled_dataset_accessor; - -template -class tile_handle +struct tile_manager_shared_data { - public: - tile_handle() = default; + struct tile_entry + { + tile_entry() : + locality_id(hpx::naming::invalid_locality_id) + { } - tile_handle(const hpx::id_type &id, std::size_t tile_index, std::size_t generation) : - ds_(id), - tile_index_(tile_index), - generation_(generation) - { } + tile_entry(hpx::id_type tile, std::uint32_t locality_id) : + tile(std::move(tile)), + locality_id(locality_id) + { } - operator mutable_tile_data() const { return get(); } + hpx::id_type tile; + std::uint32_t locality_id; + std::shared_ptr> local_data; - mutable_tile_data get() const - { - tiled_dataset_accessor ds(ds_); // TRANSITION - return ds.get_tile_data(tile_index_, generation_).get(); - } + private: + friend class hpx::serialization::access; - hpx::future> get_async() const - { - tiled_dataset_accessor ds(ds_); // TRANSITION - return ds.get_tile_data(tile_index_, generation_); - } + template + void serialize(Archive &ar, unsigned) + { + ar & tile & locality_id; + } + }; - hpx::future set_async(const mutable_tile_data &data) const - { - tiled_dataset_accessor ds(ds_); // TRANSITION - return ds.set_tile_data(tile_index_, generation_ + 1, data); - } + std::vector tiles; - private: +private: friend class hpx::serialization::access; template void serialize(Archive &ar, unsigned) { - ar & ds_ & tile_index_ & generation_; + ar & tiles; } - - // we need this instead of the actual tile_handle because per-locality caches - // reside in the accessor. - hpx::id_type ds_; - std::size_t tile_index_; - std::size_t generation_; }; template -using tiled_dataset = std::vector>>; - -template -class tiled_dataset_accessor - : public hpx::components::client_base< - tiled_dataset_accessor, - hpx::components::server::distributed_metadata_base> +struct tile_manager : hpx::components::component_base> { - using server_type = hpx::components::server::distributed_metadata_base; - using base_type = hpx::components::client_base, server_type>; - - using tile_server = server::tile_server; + tile_manager(tile_manager_shared_data &&data) : + data_(std::move(data)) + { } - struct tile_entry : server::tiled_dataset_config_data::tile_entry + mutable_tile_data get_tile_data(std::size_t tile_index, std::size_t generation) { - using base_type = server::tiled_dataset_config_data::tile_entry; - - tile_entry() = default; - - tile_entry(const hpx::id_type &part, std::uint32_t locality_id, std::size_t version) : - base_type(part, locality_id, version) - { } - - tile_entry(const base_type &base) noexcept : - base_type(base) - { } + const auto &target_tile = data_.tiles[tile_index]; - tile_entry(base_type &&base) noexcept : - base_type(HPX_MOVE(base)) - { } - - std::shared_ptr local_data; - }; + // Best is always to rely on local data + if (target_tile.local_data) + { + return target_tile.local_data->get_data(); + } - // The list of partitions belonging to this vector. - // Each partition is described by its corresponding client object, its - // size, and locality id. - using tiles_vector_type = std::vector; + // Next, try the tile cache - maybe we have current data + { + mutable_tile_data cached_data; + if (cache_.try_get(target_tile.tile.get_gid(), generation, cached_data)) + { + return cached_data; + } + } - public: - explicit tiled_dataset_accessor(const hpx::id_type &id) { connect_to(id).get(); } + hpx::chrono::high_resolution_timer timer; + auto data = hpx::async(typename tile_holder::get_data_action{}, target_tile.tile).get(); - explicit tiled_dataset_accessor(std::span> targets, - std::size_t num_tiles) - { - create(targets, num_tiles); - } + record_transmission_time(timer.elapsed_nanoseconds()); + cache_.insert(target_tile.tile.get_gid(), generation, data); - hpx::future connect_to(const hpx::id_type &id) - { - return hpx::async(server_type::get_action(), id) - .then([this, id](hpx::future &&f) -> void - { return assign_existing(id, f.get()); }); + return data; } - hpx::future> get_tile_data(std::size_t tile_index, std::size_t generation) const + hpx::future> get_tile_data_async(std::size_t tile_index, std::size_t generation) { - const auto &target_tile = tiles_[tile_index]; + const auto &target_tile = data_.tiles[tile_index]; // Best is always to rely on local data if (target_tile.local_data) @@ -306,192 +218,198 @@ class tiled_dataset_accessor // Next, try the tile cache - maybe we have current data { mutable_tile_data cached_data; - if (get_tile_cache().try_get(target_tile.tile.get_gid(), generation, cached_data)) + if (cache_.try_get(target_tile.tile.get_gid(), generation, cached_data)) { return hpx::make_ready_future(cached_data); } } - typename server::tile_server::get_data_action act; - return hpx::async(act, target_tile.tile) + return hpx::async(typename tile_holder::get_data_action{}, target_tile.tile) .then( - [generation, gid = target_tile.tile.get_gid(), timer = hpx::chrono::high_resolution_timer()]( + [this, generation, gid = target_tile.tile.get_gid(), timer = hpx::chrono::high_resolution_timer()]( hpx::future> &&f) { record_transmission_time(timer.elapsed_nanoseconds()); auto data = f.get(); - get_tile_cache().insert(gid, generation, data); + cache_.insert(gid, generation, data); return data; }); } - hpx::future> - set_tile_data(std::size_t tile_index, std::size_t generation, const mutable_tile_data &data) const + hpx::future + set_tile_data_async(std::size_t tile_index, std::size_t generation, const mutable_tile_data &data) { - const auto &target_tile = tiles_[tile_index]; + const auto &target_tile = data_.tiles[tile_index]; - auto self = base_type::get(); if (target_tile.local_data) { target_tile.local_data->set_data(data); - return hpx::make_ready_future(tile_handle{ std::move(self), tile_index, generation + 1 }); + return hpx::make_ready_future(); } // We'd lose this tile after writing it, best to put it in the cache for now - get_tile_cache().insert(target_tile.tile.get_gid(), generation, data); + cache_.insert(target_tile.tile.get_gid(), generation, data); - typename server::tile_server::set_data_action act; - return hpx::async(act, target_tile.tile, data) - .then([self = std::move(self), tile_index, generation](const hpx::future &) - { return tile_handle{ std::move(self), tile_index, generation + 1 }; }); + typename tile_holder::set_data_action act; + return hpx::async(act, target_tile.tile, data); } - // TRANSITION - tiled_dataset to_dataset() + private: + tile_manager_shared_data data_; + tile_cache cache_; +}; + +} // namespace server + +// DECLARATION macros (use in a single header) + +#define GPRAT_REGISTER_TILE_HOLDER_DECLARATION_IMPL(type, name) \ + HPX_REGISTER_ACTION_DECLARATION(type::get_data_action, HPX_PP_CAT(_tile_holder_get_data_action_, name)) \ + HPX_REGISTER_ACTION_DECLARATION(type::set_data_action, HPX_PP_CAT(_tile_holder_set_data_action_, name)) \ + /**/ + +#define GPRAT_REGISTER_TILED_DATASET_DECLARATION(type, name) \ + typedef ::GPRAT_NS::server::tile_holder HPX_PP_CAT(_server_tile_holder_, HPX_PP_CAT(type, name)); \ + GPRAT_REGISTER_TILE_HOLDER_DECLARATION_IMPL(HPX_PP_CAT(_server_tile_holder_, HPX_PP_CAT(type, name)), name) + +// REGISTRATION macros (use in a single .cpp file) + +#define GPRAT_REGISTER_TILE_HOLDER_IMPL(type, name) \ + HPX_REGISTER_ACTION(type::get_data_action, HPX_PP_CAT(_tile_holder_get_data_action_, name)) \ + HPX_REGISTER_ACTION(type::set_data_action, HPX_PP_CAT(_tile_holder_set_data_action_, name)) \ + typedef ::hpx::components::component HPX_PP_CAT(_server_tile_holder_component_, name); \ + HPX_REGISTER_COMPONENT(HPX_PP_CAT(_server_tile_holder_component_, name)) \ + /**/ + +#define GPRAT_REGISTER_TILE_MANAGER_IMPL(type, name) \ + typedef ::hpx::components::component HPX_PP_CAT(_server_tile_manager_component_, name); \ + HPX_REGISTER_COMPONENT(HPX_PP_CAT(_server_tile_manager_component_, name)) \ + /**/ + +#define GPRAT_REGISTER_TILED_DATASET(type, name) \ + typedef ::GPRAT_NS::server::tile_holder HPX_PP_CAT(_server_tile_holder_, HPX_PP_CAT(type, name)); \ + GPRAT_REGISTER_TILE_HOLDER_IMPL(HPX_PP_CAT(_server_tile_holder_, HPX_PP_CAT(type, name)), name) \ + typedef ::GPRAT_NS::server::tile_manager HPX_PP_CAT(_server_tile_manager_, HPX_PP_CAT(type, name)); \ + GPRAT_REGISTER_TILE_MANAGER_IMPL(HPX_PP_CAT(_server_tile_manager_, HPX_PP_CAT(type, name)), name) + +template +class tiled_dataset_accessor; + +template +class tile_handle +{ + public: + tile_handle() = default; + + tile_handle(std::vector managers, std::size_t tile_index, std::size_t generation) : + managers_(std::move(managers)), + tile_index_(tile_index), + generation_(generation) + { } + + operator mutable_tile_data() const { return get(); } + + mutable_tile_data get() const { return get_local_manager()->get_tile_data(tile_index_, generation_); } + + hpx::future> get_async() const { - tiled_dataset result; - result.reserve(tiles_.size()); - for (std::size_t i = 0; i < tiles_.size(); ++i) - { - result.emplace_back(hpx::make_ready_future(tile_handle{ base_type::get(), i, tiles_[i].generation })); - } - return result; + return get_local_manager()->get_tile_data_async(tile_index_, generation_); } - private: - void assign_existing(const hpx::id_type &id, server::tiled_dataset_config_data &&config) + hpx::future set_async(const mutable_tile_data &data) const { - tiles_.clear(); - tiles_.insert(tiles_.end(), config.tiles.begin(), config.tiles.end()); + return get_local_manager() + ->set_tile_data_async(tile_index_, generation_ + 1, data) + .then( + [self = *this](hpx::future &&) mutable + { + ++self.generation_; + return self; + }); + } - const auto here = hpx::get_locality_id(); - for (auto &tile : tiles_) - { - if (tile.locality_id == here && !tile.local_data) - { - tile.local_data = hpx::get_ptr(hpx::launch::sync, tile.tile); - } - } + private: + friend class hpx::serialization::access; - return base_type::reset(id); + template + void serialize(Archive &ar, unsigned) + { + ar & managers_ & tile_index_ & generation_; } - void create(std::span> targets, std::size_t num_tiles) + std::shared_ptr> get_local_manager() const { - std::vector>> objs; - objs.reserve(targets.size()); - for (const auto &target : targets) - { - objs.emplace_back(hpx::components::bulk_create_async(target.first, target.second)); - } - const auto here = hpx::get_locality_id(); - tiles_.resize(num_tiles); - - std::size_t l = 0; - for (std::size_t i = 0; i < targets.size(); ++i) + for (const auto &id : managers_) { - const auto locality = hpx::naming::get_locality_id_from_id(targets[i].first); - for (const hpx::id_type &id : objs[i].get()) + if (here == hpx::naming::get_locality_id_from_id(id)) { - tiles_[l] = tile_entry(id, locality, 0); - - if (locality == here) - { - tiles_[l].local_data = hpx::get_ptr(hpx::launch::sync, id); - } - - if (++l == num_tiles) - { - break; - } + return hpx::get_ptr>(hpx::launch::sync, id); } } - HPX_ASSERT(l == num_tiles); - std::vector data{ tiles_.begin(), tiles_.end() }; - base_type::reset( - hpx::new_>( - hpx::find_here(), server::tiled_dataset_config_data{ std::move(data) })); + throw std::runtime_error("This locality is not known"); } - tiles_vector_type tiles_; + // TODO: It would be best if the caller could give us the right manager already, + // but since the amount of localities is somewhat limited, this will do for now. + std::vector managers_; + std::size_t tile_index_; + std::size_t generation_; }; -// partitioned_vector_partition => tile_handle? -// server::partitioned_vector => tile_server? - -/*template -class tiled_dataset -{ - - - hpx::future>& operator[](std::size_t index) - { - return - } - // operator[] => future[tile_reference]& -};*/ - -// OLD: -/* -/////////////////////////////////////////////////////////////////////////////// -// This is a client side helper class allowing to hide some of the tedious -// boilerplate while referencing a remote partition. +template +using tiled_dataset = std::vector>>; template -struct tile_handle : hpx::components::client_base, server::tile_server> +tiled_dataset +create_tiled_dataset(std::span> targets, std::size_t num_tiles) { - using base_type = hpx::components::client_base>; - - tile_handle() = default; + using data_type = server::tile_manager_shared_data; - // Create new component on locality 'where' and initialize the held data - tile_handle(const hpx::id_type &where, const mutable_tile_data &data) : - base_type(hpx::new_>(where, data)) - { } - - // Create new component on locality 'where' and initialize the held data - template - requires hpx::traits::is_distribution_policy_v tile_handle(const T &policy, const mutable_tile_data &data) : - base_type(hpx::new_>(policy, data)) - { } + // First, create the actual tile data holders + std::vector>> holders; + holders.reserve(targets.size()); + for (const auto &target : targets) + { + holders.emplace_back(hpx::components::bulk_create_async>(target.first, target.second)); + } - // Attach a future representing a (possibly remote) partition. - // ReSharper disable once CppNonExplicitConvertingConstructor - tile_handle(hpx::future &&id) noexcept : - base_type(std::move(id)) - { } + // Next we prepare our shared data for the manager components + data_type manager_data; + manager_data.tiles.resize(num_tiles); - // Unwrap a future (a tile_handle already is a future to the - // id of the referenced object, thus unwrapping accesses this inner future). - // ReSharper disable once CppNonExplicitConvertingConstructor - tile_handle(hpx::future &&c) noexcept : - base_type(std::move(c)) - { } + std::size_t l = 0; + for (std::size_t i = 0; i < targets.size(); ++i) + { + const auto locality = hpx::naming::get_locality_id_from_id(targets[i].first); + for (hpx::id_type &id : holders[i].get()) + { + manager_data.tiles[l++] = data_type::tile_entry(std::move(id), locality); + if (l == num_tiles) + { + break; + } + } + } + HPX_ASSERT(l == num_tiles); - /////////////////////////////////////////////////////////////////////////// - // tile's are immutable for now, meaning we can cache them on first use. - [[nodiscard]] hpx::future> get_data() const + // Now we move on to the manager components + std::vector managers; + managers.reserve(targets.size()); + for (const auto &target : targets) { - typename server::tile_server::get_data_action act; - return hpx::async(act, base_type::get_id()); + managers.emplace_back(hpx::components::create>(target.first, manager_data)); } - [[nodiscard]] hpx::future set_data(const mutable_tile_data &data) + // Finally, we create our fat tile_handles + tiled_dataset tiles; + tiles.reserve(num_tiles); + for (std::size_t i = 0; i < num_tiles; ++i) { - typename server::tile_server::set_data_action act; - return hpx::async(act, base_type::get_id(), data); + tiles.push_back(hpx::make_ready_future(tile_handle{managers, i, 0})); } -}; -*/ + return tiles; +} GPRAT_NS_END - -/* -// serialization of partitioned_vector requires special handling -template -struct hpx::traits::needs_reference_semantics> - : std::true_type -{ -};*/ diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index 8097e351..d77ff0e5 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -65,7 +65,7 @@ void right_looking_cholesky_tiled(Scheduler &sched, Tiles &tiles, std::size_t N, // TRSM: Solve X * L^T = A tiles[m * n_tiles + k] = detail::named_dataflow( sched, - cholesky_TRSM(sched, k, m), + cholesky_TRSM(sched, n_tiles, k, m), "cholesky_tiled", tiles[k * n_tiles + k], tiles[m * n_tiles + k], From 49171718045ae22c112a7783b18fe9e42fc91901 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sat, 26 Jul 2025 01:25:24 +0200 Subject: [PATCH 49/56] fix(examples): Don't create-then-overwrite metadata tiles --- examples/distributed/src/distributed_tile.hpp | 11 ++++------- 1 file changed, 4 insertions(+), 7 deletions(-) diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index 5f2446f5..4cb01b6d 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -365,8 +365,6 @@ template tiled_dataset create_tiled_dataset(std::span> targets, std::size_t num_tiles) { - using data_type = server::tile_manager_shared_data; - // First, create the actual tile data holders std::vector>> holders; holders.reserve(targets.size()); @@ -376,17 +374,16 @@ create_tiled_dataset(std::span> targe } // Next we prepare our shared data for the manager components - data_type manager_data; - manager_data.tiles.resize(num_tiles); + server::tile_manager_shared_data manager_data; + manager_data.tiles.reserve(num_tiles); - std::size_t l = 0; for (std::size_t i = 0; i < targets.size(); ++i) { const auto locality = hpx::naming::get_locality_id_from_id(targets[i].first); for (hpx::id_type &id : holders[i].get()) { - manager_data.tiles[l++] = data_type::tile_entry(std::move(id), locality); - if (l == num_tiles) + manager_data.tiles.emplace_back(std::move(id), locality); + if (manager_data.tiles.size() == num_tiles) { break; } From 1357559350a9a9bfed46fb4c1bac7bbb6cdebf16 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sat, 26 Jul 2025 05:02:46 +0200 Subject: [PATCH 50/56] fix(examples): Use shared_mutex for tile_holder --- examples/distributed/src/distributed_tile.hpp | 21 ++++++++++++------- 1 file changed, 14 insertions(+), 7 deletions(-) diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index 4cb01b6d..d914e6c8 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -105,7 +105,7 @@ namespace server * requires T to be serializable by HPX. */ template -struct tile_holder : hpx::components::locking_hook>> +struct tile_holder : hpx::components::component_base> { tile_holder() { track_tile_server_allocation(0); } @@ -117,9 +117,17 @@ struct tile_holder : hpx::components::locking_hook get_data() const { return data_; } + [[nodiscard]] mutable_tile_data get_data() const + { + std::shared_lock lock(mutex_); + return data_; + } - void set_data(const mutable_tile_data &data) { data_ = data; } + void set_data(const mutable_tile_data &data) + { + std::unique_lock lock(mutex_); + data_ = data; + } // Every member function that has to be invoked remotely needs to be // wrapped into a component action. @@ -127,6 +135,7 @@ struct tile_holder : hpx::components::locking_hook data_; }; @@ -173,7 +182,7 @@ struct tile_manager_shared_data template struct tile_manager : hpx::components::component_base> { - tile_manager(tile_manager_shared_data &&data) : + explicit tile_manager(tile_manager_shared_data &&data) : data_(std::move(data)) { } @@ -250,8 +259,7 @@ struct tile_manager : hpx::components::component_base> // We'd lose this tile after writing it, best to put it in the cache for now cache_.insert(target_tile.tile.get_gid(), generation, data); - typename tile_holder::set_data_action act; - return hpx::async(act, target_tile.tile, data); + return hpx::async(typename tile_holder::set_data_action{}, target_tile.tile, data); } private: @@ -389,7 +397,6 @@ create_tiled_dataset(std::span> targe } } } - HPX_ASSERT(l == num_tiles); // Now we move on to the manager components std::vector managers; From f97ba0e2c4ef92471ba66d349da19f9070a6700a Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sat, 26 Jul 2025 05:12:16 +0200 Subject: [PATCH 51/56] fix(examples): Populate local tile pointers --- examples/distributed/src/distributed_tile.hpp | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index d914e6c8..09ca9ecb 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -135,7 +135,7 @@ struct tile_holder : hpx::components::component_base> HPX_DEFINE_COMPONENT_DIRECT_ACTION(tile_holder, set_data) private: - hpx::shared_mutex mutex_; + mutable hpx::shared_mutex mutex_; mutable_tile_data data_; }; @@ -184,7 +184,16 @@ struct tile_manager : hpx::components::component_base> { explicit tile_manager(tile_manager_shared_data &&data) : data_(std::move(data)) - { } + { + const auto here = hpx::get_locality_id(); + for (auto& tile : data_.tiles) + { + if (tile.locality_id == here) + { + tile.local_data = hpx::get_ptr>(hpx::launch::sync, tile.tile); + } + } + } mutable_tile_data get_tile_data(std::size_t tile_index, std::size_t generation) { @@ -381,7 +390,7 @@ create_tiled_dataset(std::span> targe holders.emplace_back(hpx::components::bulk_create_async>(target.first, target.second)); } - // Next we prepare our shared data for the manager components + // Next, we prepare our shared data for the manager components server::tile_manager_shared_data manager_data; manager_data.tiles.reserve(num_tiles); From 885f0c3594ccf533829069eb9c8349851d225e48 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sat, 26 Jul 2025 21:26:30 +0200 Subject: [PATCH 52/56] chore(examples): Add more counters --- examples/distributed/src/distributed_tile.cpp | 56 ++++++++----------- examples/distributed/src/distributed_tile.hpp | 38 ++++++++----- 2 files changed, 46 insertions(+), 48 deletions(-) diff --git a/examples/distributed/src/distributed_tile.cpp b/examples/distributed/src/distributed_tile.cpp index f4acfd2b..8807c1d7 100644 --- a/examples/distributed/src/distributed_tile.cpp +++ b/examples/distributed/src/distributed_tile.cpp @@ -15,21 +15,8 @@ hpx::util::cache::statistics::local_full_statistics &get_global_statistics() } // namespace detail -struct tile_cache_counters -{ - // XXX: you can do this with templates, but it's quite a bit more complicated -#define GPRAT_MAKE_STATISTICS_ACCESSOR(name) \ - static std::uint64_t get_cache_##name(bool reset) { return detail::get_global_statistics().name(reset); } - - GPRAT_MAKE_STATISTICS_ACCESSOR(hits); - GPRAT_MAKE_STATISTICS_ACCESSOR(misses); - GPRAT_MAKE_STATISTICS_ACCESSOR(evictions); - GPRAT_MAKE_STATISTICS_ACCESSOR(insertions); - -#undef GPRAT_MAKE_STATISTICS_ACCESSOR -}; - std::atomic tile_transmission_time(0); +std::atomic tile_transmission_count(0); std::atomic tile_data_allocations(0); std::atomic tile_data_deallocations(0); std::atomic tile_server_allocations(0); @@ -38,6 +25,7 @@ std::atomic tile_server_deallocations(0); void record_transmission_time(std::int64_t elapsed_ns) { HPX_ASSERT(elapsed_ns >= 0); + tile_transmission_count += 1; if (elapsed_ns > 0) { tile_transmission_time += static_cast(elapsed_ns); @@ -51,6 +39,7 @@ void track_tile_server_deallocation(std::size_t /*size*/) { tile_server_dealloca #define GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(name) \ std::uint64_t get_##name(bool reset) { return hpx::util::get_and_reset_value(name, reset); } +GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_transmission_count) GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_transmission_time) GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_server_allocations) GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_server_deallocations) @@ -59,30 +48,29 @@ GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_server_deallocations) void register_distributed_tile_counters() { + // XXX: you can do this with templates, but it's quite a bit more complicated +#define GPRAT_MAKE_STATISTICS_ACCESSOR(name, stats_expr) \ + hpx::performance_counters::install_counter_type( \ + name, \ + [](bool reset) { return (stats_expr) (reset); }, \ + #stats_expr, \ + "", \ + hpx::performance_counters::counter_type::monotonically_increasing) + + GPRAT_MAKE_STATISTICS_ACCESSOR("/gprat/tile_cache/hits", detail::get_global_statistics().hits); + GPRAT_MAKE_STATISTICS_ACCESSOR("/gprat/tile_cache/misses", detail::get_global_statistics().misses); + GPRAT_MAKE_STATISTICS_ACCESSOR("/gprat/tile_cache/evictions", detail::get_global_statistics().evictions); + GPRAT_MAKE_STATISTICS_ACCESSOR("/gprat/tile_cache/insertions", detail::get_global_statistics().insertions); + +#undef GPRAT_MAKE_STATISTICS_ACCESSOR + hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/hits", - &tile_cache_counters::get_cache_hits, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/misses", - &tile_cache_counters::get_cache_misses, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/evictions", - &tile_cache_counters::get_cache_evictions, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/insertions", - &tile_cache_counters::get_cache_insertions, + "/gprat/tile_cache/transmission_count", + &get_tile_transmission_time, "", "", hpx::performance_counters::counter_type::monotonically_increasing); + hpx::performance_counters::install_counter_type( "/gprat/tile_cache/transmission_time", &get_tile_transmission_time, diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index 09ca9ecb..a6ce9b74 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -34,7 +34,10 @@ class global_full_statistics using update_on_exit = hpx::util::cache::statistics::local_full_statistics::update_on_exit; // ReSharper disable once CppNonExplicitConversionOperator - operator hpx::util::cache::statistics::local_full_statistics &() const { return get_global_statistics(); } + operator hpx::util::cache::statistics::local_full_statistics &() const + { + return get_global_statistics(); + } void got_hit() noexcept { get_global_statistics().got_hit(); } @@ -61,18 +64,24 @@ class tile_cache bool try_get(const hpx::naming::gid_type &key, std::size_t generation, mutable_tile_data &cached_data) { std::lock_guard g(mutex_); - hpx::naming::gid_type unused; + entry e; - if (cache_.get_entry(key, unused, e)) { - if (e.generation == generation) + hpx::naming::gid_type unused; + if (!cache_.get_entry(key, unused, e)) { - cached_data = e.data; - return true; + return false; } - // Erase the obsolete entry - cache_.erase([&](const auto &p) { return p.first == key; }); } + + if (e.generation == generation) + { + cached_data = e.data; + return true; + } + + // Erase the obsolete entry + cache_.erase([&](const auto &p) { return p.first == key; }); return false; } @@ -169,7 +178,7 @@ struct tile_manager_shared_data std::vector tiles; -private: + private: friend class hpx::serialization::access; template @@ -186,7 +195,7 @@ struct tile_manager : hpx::components::component_base> data_(std::move(data)) { const auto here = hpx::get_locality_id(); - for (auto& tile : data_.tiles) + for (auto &tile : data_.tiles) { if (tile.locality_id == here) { @@ -324,7 +333,8 @@ class tile_handle generation_(generation) { } - operator mutable_tile_data() const { return get(); } + // ReSharper disable once CppNonExplicitConversionOperator + operator mutable_tile_data() const { return get(); } // NOLINT(*-explicit-constructor) mutable_tile_data get() const { return get_local_manager()->get_tile_data(tile_index_, generation_); } @@ -371,8 +381,8 @@ class tile_handle // TODO: It would be best if the caller could give us the right manager already, // but since the amount of localities is somewhat limited, this will do for now. std::vector managers_; - std::size_t tile_index_; - std::size_t generation_; + std::size_t tile_index_ = 0; + std::size_t generation_ = 0; }; template @@ -420,7 +430,7 @@ create_tiled_dataset(std::span> targe tiles.reserve(num_tiles); for (std::size_t i = 0; i < num_tiles; ++i) { - tiles.push_back(hpx::make_ready_future(tile_handle{managers, i, 0})); + tiles.push_back(hpx::make_ready_future(tile_handle{ managers, i, 0 })); } return tiles; } From a9846917c8514b0a793c53559938650b9a3cd3d6 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sat, 26 Jul 2025 21:38:56 +0200 Subject: [PATCH 53/56] fix(examples): Fix HPX 1.11 support --- examples/distributed/src/distributed_tile.hpp | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index a6ce9b74..db69517a 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -34,10 +34,7 @@ class global_full_statistics using update_on_exit = hpx::util::cache::statistics::local_full_statistics::update_on_exit; // ReSharper disable once CppNonExplicitConversionOperator - operator hpx::util::cache::statistics::local_full_statistics &() const - { - return get_global_statistics(); - } + operator hpx::util::cache::statistics::local_full_statistics &() const { return get_global_statistics(); } void got_hit() noexcept { get_global_statistics().got_hit(); } @@ -397,7 +394,12 @@ create_tiled_dataset(std::span> targe holders.reserve(targets.size()); for (const auto &target : targets) { +#if (HPX_VERSION_FULL >= 0x011100) + holders.emplace_back( + hpx::components::bulk_create_async>(target.first, target.second)); +#else holders.emplace_back(hpx::components::bulk_create_async>(target.first, target.second)); +#endif } // Next, we prepare our shared data for the manager components From 38d0c9966e9dd84d8d2ccd5285341596a320f3d4 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sun, 21 Sep 2025 00:10:20 +0200 Subject: [PATCH 54/56] wip: Final version for tests --- core/include/gprat/detail/actions.hpp | 119 ++++ core/include/gprat/tile_data.hpp | 11 +- examples/distributed/CMakeLists.txt | 11 +- examples/distributed/src/distributed_blas.cpp | 120 ++-- examples/distributed/src/distributed_blas.hpp | 35 +- .../distributed/src/distributed_cholesky.hpp | 316 ++++++--- examples/distributed/src/distributed_tile.cpp | 2 +- examples/distributed/src/distributed_tile.hpp | 78 ++- examples/distributed/src/main.cpp | 603 ++++++++++++++---- examples/distributed/src/scheduling.cpp | 67 +- examples/distributed/src/scheduling.hpp | 121 +--- test/src/output_correctness.cpp | 30 - test/src/test_data.hpp | 32 + 13 files changed, 1057 insertions(+), 488 deletions(-) create mode 100644 core/include/gprat/detail/actions.hpp diff --git a/core/include/gprat/detail/actions.hpp b/core/include/gprat/detail/actions.hpp new file mode 100644 index 00000000..cc047fbb --- /dev/null +++ b/core/include/gprat/detail/actions.hpp @@ -0,0 +1,119 @@ +#ifndef GPRAT_DETAIL_ACTIONS_HPP +#define GPRAT_DETAIL_ACTIONS_HPP + +#pragma once + +#include "gprat/detail/config.hpp" + +#include +#include +#include + +GPRAT_NS_BEGIN + +/// @brief This template provides access to a function F's associated HPX action and related metadata. +/// +/// Users can use this template to access the previously declared HPX plain (and optionally direct) action. +/// This way we get singleton-like semantics for free, there is always only one plain action associated with +/// a Callable value F. +template +struct plain_action_for; + +#define GPRAT_DECLARE_PLAIN_ACTION_FOR(local_function, action, friendly_name) \ + template <> \ + struct plain_action_for \ + { \ + using action_type = action; \ + constexpr static std::string_view name = friendly_name; \ + } + +#define GPRAT_DEFINE_PLAIN_ACTION_FOR(local_function) + +// ============================================================= +// distributed action-based scheduling + +struct tiled_scheduler_distributed +{ + /// @brief Create a new scheduler that targets all localities. + tiled_scheduler_distributed() : + localities_(hpx::find_all_localities()) + { + // ctor + } + + /// @brief Create a new scheduler that targets the given localities. + explicit tiled_scheduler_distributed(std::vector in_localities) : + localities_(std::move(in_localities)) + { + // ctor + } + + std::vector localities_; +}; + +namespace detail +{ +// HPX does not auto-collapse future chains in their async(), dataflow(), ... functions. +// This usually works fine, but we require shared_futures most of the time. +// Unfortunately, C++ will not do two-step conversions for us (future> -> future -> shared_future). +// see: https://github.com/STEllAR-GROUP/hpx/issues/3758 +template +hpx::future collapse(hpx::future> &&fut) +{ + return { std::move(fut) }; +} + +template +hpx::future collapse(hpx::future &&fut) +{ + return std::move(fut); +} + +template +decltype(auto) +named_make_tile(const tiled_scheduler_distributed &sched, std::size_t on, const char *name, Args &&...args) +{ + hpx::threads::thread_schedule_hint hint; + hint.sharing_mode(hpx::threads::thread_sharing_hint::do_not_combine_tasks + | hpx::threads::thread_sharing_hint::do_not_share_function); + decltype(auto) policy = hpx::execution::experimental::with_hint(hpx::launch::async, hint) | hpx::launch::deferred; + return collapse(hpx::dataflow( + policy, + hpx::annotated_function(hpx::unwrapping(typename plain_action_for::action_type{}), name), + hpx::find_here(), // sched.localities_[on], + std::forward(args)...)); +} + +template +decltype(auto) +named_dataflow(const tiled_scheduler_distributed &sched, std::size_t on, const char *name, Args &&...args) +{ + hpx::threads::thread_schedule_hint hint; + hint.sharing_mode(hpx::threads::thread_sharing_hint::do_not_combine_tasks + | hpx::threads::thread_sharing_hint::do_not_share_function); + decltype(auto) policy = hpx::execution::experimental::with_hint(hpx::launch::async, hint) | hpx::launch::deferred; + return collapse(hpx::dataflow( + policy, + hpx::annotated_function(hpx::unwrapping(typename plain_action_for::action_type{}), name), + sched.localities_[on], + std::forward(args)...)); +} + +template +decltype(auto) named_async(const tiled_scheduler_distributed &sched, std::size_t on, const char *name, Args &&...args) +{ + hpx::threads::thread_schedule_hint hint; + hint.sharing_mode(hpx::threads::thread_sharing_hint::do_not_combine_tasks + | hpx::threads::thread_sharing_hint::do_not_share_function); + decltype(auto) policy = hpx::execution::experimental::with_hint(hpx::launch::async, hint) | hpx::launch::deferred; + return hpx::async(policy, + hpx::annotated_function(policy, typename plain_action_for::action_type{}, name), + sched.localities_[on], + std::forward(args)...); +} + +} // namespace detail + +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/tile_data.hpp b/core/include/gprat/tile_data.hpp index a2615ad8..96fd3f3c 100644 --- a/core/include/gprat/tile_data.hpp +++ b/core/include/gprat/tile_data.hpp @@ -102,7 +102,12 @@ class const_tile_data hold_reference(base.cpu_data_)) // keep referenced tile_data alive { } - [[nodiscard]] const T *data() const noexcept { return cpu_data_.data(); } + [[nodiscard]] const T *data() const + { + if (!cpu_data_.data()) + throw std::runtime_error("no data"); + return cpu_data_.data(); + } [[nodiscard]] std::size_t size() const noexcept { return cpu_data_.size(); } @@ -150,7 +155,9 @@ class mutable_tile_data : public const_tile_data public: using const_tile_data::const_tile_data; - [[nodiscard]] T *data() const noexcept { return const_cast(this->cpu_data_.data()); } + [[nodiscard]] T *data() const { + if (!this->cpu_data_.data()) + throw std::runtime_error("no data");return const_cast(this->cpu_data_.data()); } [[nodiscard]] T *begin() const noexcept { return const_cast(this->cpu_data_.data()); } diff --git a/examples/distributed/CMakeLists.txt b/examples/distributed/CMakeLists.txt index 3c0275b2..8abff6e8 100644 --- a/examples/distributed/CMakeLists.txt +++ b/examples/distributed/CMakeLists.txt @@ -2,8 +2,17 @@ add_executable(gprat_distributed src/main.cpp src/distributed_blas.cpp src/distributed_tile.cpp) target_compile_features(gprat_distributed PUBLIC cxx_std_20) +include(FetchContent) + +FetchContent_Declare( + Catch2 + GIT_REPOSITORY https://github.com/catchorg/Catch2.git + GIT_TAG v3.8.0) + +FetchContent_MakeAvailable(Catch2) + find_package(Boost REQUIRED) -target_link_libraries(gprat_distributed PUBLIC GPRat::core HPX::hpx +target_link_libraries(gprat_distributed PUBLIC GPRat::core HPX::hpx Catch2::Catch2 Boost::boost) set_target_properties(gprat_distributed PROPERTIES VS_DEBUGGER_WORKING_DIRECTORY diff --git a/examples/distributed/src/distributed_blas.cpp b/examples/distributed/src/distributed_blas.cpp index 84665a24..0e0af8b4 100644 --- a/examples/distributed/src/distributed_blas.cpp +++ b/examples/distributed/src/distributed_blas.cpp @@ -2,13 +2,17 @@ #include "gprat/cpu/adapter_cblas_fp64.hpp" -#include #include HPX_REGISTER_ACTION(GPRAT_NS::potrf_distributed_action); HPX_REGISTER_ACTION(GPRAT_NS::trsm_distributed_action); HPX_REGISTER_ACTION(GPRAT_NS::syrk_distributed_action); HPX_REGISTER_ACTION(GPRAT_NS::gemm_distributed_action); +HPX_REGISTER_ACTION(GPRAT_NS::trsv_distributed_action); +HPX_REGISTER_ACTION(GPRAT_NS::gemv_distributed_action); +HPX_REGISTER_ACTION(GPRAT_NS::dot_diag_syrk_distributed_action); +HPX_REGISTER_ACTION(GPRAT_NS::dot_diag_gemm_distributed_action); +HPX_REGISTER_ACTION(GPRAT_NS::axpy_distributed_action); GPRAT_NS_BEGIN @@ -16,16 +20,17 @@ GPRAT_DEFINE_PLAIN_ACTION_FOR(&potrf); GPRAT_DEFINE_PLAIN_ACTION_FOR(&trsm); GPRAT_DEFINE_PLAIN_ACTION_FOR(&syrk); GPRAT_DEFINE_PLAIN_ACTION_FOR(&gemm); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&trsv); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&gemv); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&dot_diag_syrk); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&dot_diag_gemm); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&axpy); hpx::future> potrf_distributed(const tile_handle &A, int N) { return hpx::dataflow( hpx::launch::async, - [A, N](hpx::future> &&tile) - { - GPRAT_TIME_PLAIN_ACTION(potrf); - return A.set_async(potrf(tile.get(), N)); - }, + [A, N](hpx::future> &&tile) { return A.set_async(potrf(tile.get(), N)); }, A.get_async()); } @@ -41,10 +46,7 @@ hpx::future> trsm_distributed( hpx::launch::async, [A, N, M, transpose_L, side_L]( hpx::future> &&Ld, hpx::future> &&Ad) - { - GPRAT_TIME_PLAIN_ACTION(trsm); - return A.set_async(trsm(Ld.get(), Ad.get(), N, M, transpose_L, side_L)); - }, + { return A.set_async(trsm(Ld.get(), Ad.get(), N, M, transpose_L, side_L)); }, L.get_async(), A.get_async()); } @@ -54,10 +56,7 @@ hpx::future> syrk_distributed(const tile_handle &A, return hpx::dataflow( hpx::launch::async, [A, N](hpx::future> &&Ad, hpx::future> &&Bd) - { - GPRAT_TIME_PLAIN_ACTION(syrk); - return A.set_async(syrk(Ad.get(), Bd.get(), N)); - }, + { return A.set_async(syrk(Ad.get(), Bd.get(), N)); }, A.get_async(), B.get_async()); } @@ -77,41 +76,76 @@ hpx::future> gemm_distributed( [C, N, M, K, transpose_A, transpose_B](hpx::future> &&Ad, hpx::future> &&Bd, hpx::future> &&Cd) - { - GPRAT_TIME_PLAIN_ACTION(gemm); - return C.set_async(gemm(Ad.get(), Bd.get(), Cd.get(), N, M, K, transpose_A, transpose_B)); - }, + { return C.set_async(gemm(Ad.get(), Bd.get(), Cd.get(), N, M, K, transpose_A, transpose_B)); }, A.get_async(), B.get_async(), C.get_async()); } -void register_distributed_blas_counters() +hpx::future> +trsv_distributed(const tile_handle &L, const tile_handle &a, int N, BLAS_TRANSPOSE transpose_L) { - hpx::performance_counters::install_counter_type( - "/gprat/potrf/time", - get_and_reset_plain_action_elapsed<&potrf>, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/trsm/time", - get_and_reset_plain_action_elapsed<&trsm>, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/syrk/time", - get_and_reset_plain_action_elapsed<&syrk>, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/gemm/time", - get_and_reset_plain_action_elapsed<&gemm>, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); + return hpx::dataflow( + hpx::launch::async, + [a, N, transpose_L](hpx::future> &&Ld, hpx::future> &&ad) + { return a.set_async(trsv(Ld.get(), ad.get(), N, transpose_L)); }, + L.get_async(), + a.get_async()); +} + +hpx::future> gemv_distributed( + const tile_handle &A, + const tile_handle &a, + const tile_handle &b, + int N, + int M, + BLAS_ALPHA alpha, + BLAS_TRANSPOSE transpose_A) +{ + return hpx::dataflow( + hpx::launch::async, + [b, N, M, alpha, transpose_A](hpx::future> &&Ad, + hpx::future> &&ad, + hpx::future> &&bd) + { return b.set_async(gemv(Ad.get(), ad.get(), bd.get(), N, M, alpha, transpose_A)); }, + A.get_async(), + a.get_async(), + b.get_async()); +} + +hpx::future> +dot_diag_syrk_distributed(const tile_handle &A, const tile_handle &r, int N, int M) +{ + return hpx::dataflow( + hpx::launch::async, + [r, N, M](hpx::future> &&Ad, hpx::future> &&rd) + { return r.set_async(dot_diag_syrk(Ad.get(), rd.get(), N, M)); }, + A.get_async(), + r.get_async()); +} + +hpx::future> dot_diag_gemm_distributed( + const tile_handle &A, const tile_handle &B, const tile_handle &r, int N, int M) +{ + return hpx::dataflow( + hpx::launch::async, + [r, N, M](hpx::future> &&Ad, + hpx::future> &&Bd, + hpx::future> &&rd) + { return r.set_async(dot_diag_gemm(Ad.get(), Bd.get(), rd.get(), N, M)); }, + A.get_async(), + B.get_async(), + r.get_async()); +} + +hpx::future> axpy_distributed(const tile_handle &y, const tile_handle &x, int N) +{ + return hpx::dataflow( + hpx::launch::async, + [y, N](hpx::future> &&yd, hpx::future> &&xd) + { return y.set_async(axpy(yd.get(), xd.get(), N)); }, + y.get_async(), + x.get_async()); } GPRAT_NS_END diff --git a/examples/distributed/src/distributed_blas.hpp b/examples/distributed/src/distributed_blas.hpp index 493cf548..7b19283b 100644 --- a/examples/distributed/src/distributed_blas.hpp +++ b/examples/distributed/src/distributed_blas.hpp @@ -29,17 +29,43 @@ hpx::future> gemm_distributed( BLAS_TRANSPOSE transpose_A, BLAS_TRANSPOSE transpose_B); +hpx::future> +trsv_distributed(const tile_handle &L, const tile_handle &a, int N, BLAS_TRANSPOSE transpose_L); +hpx::future> gemv_distributed( + const tile_handle &A, + const tile_handle &a, + const tile_handle &b, + int N, + int M, + BLAS_ALPHA alpha, + BLAS_TRANSPOSE transpose_A); + +hpx::future> +dot_diag_syrk_distributed(const tile_handle &A, const tile_handle &r, int N, int M); +hpx::future> dot_diag_gemm_distributed( +const tile_handle &A, const tile_handle &B, const tile_handle &r, int N, int M); +hpx::future> axpy_distributed( + const tile_handle &y, const tile_handle &x, int N); + HPX_DEFINE_PLAIN_DIRECT_ACTION(potrf_distributed); HPX_DEFINE_PLAIN_DIRECT_ACTION(trsm_distributed); HPX_DEFINE_PLAIN_DIRECT_ACTION(syrk_distributed); HPX_DEFINE_PLAIN_DIRECT_ACTION(gemm_distributed); +HPX_DEFINE_PLAIN_DIRECT_ACTION(trsv_distributed); +HPX_DEFINE_PLAIN_DIRECT_ACTION(gemv_distributed); +HPX_DEFINE_PLAIN_DIRECT_ACTION(dot_diag_syrk_distributed); +HPX_DEFINE_PLAIN_DIRECT_ACTION(dot_diag_gemm_distributed); +HPX_DEFINE_PLAIN_DIRECT_ACTION(axpy_distributed); GPRAT_DECLARE_PLAIN_ACTION_FOR(&potrf, potrf_distributed_action, "POTRF"); GPRAT_DECLARE_PLAIN_ACTION_FOR(&trsm, trsm_distributed_action, "TRSM"); GPRAT_DECLARE_PLAIN_ACTION_FOR(&syrk, syrk_distributed_action, "SYRK"); GPRAT_DECLARE_PLAIN_ACTION_FOR(&gemm, gemm_distributed_action, "GEMM"); - -void register_distributed_blas_counters(); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&trsv, trsv_distributed_action, "TRSV"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&gemv, gemv_distributed_action, "GEMV"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&dot_diag_syrk, dot_diag_syrk_distributed_action, "dot diag(SYRK)"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&dot_diag_gemm, dot_diag_gemm_distributed_action, "dot diag(GEMM)"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&axpy, axpy_distributed_action, "axpy"); GPRAT_NS_END @@ -47,3 +73,8 @@ HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::potrf_distributed_action); HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::trsm_distributed_action); HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::syrk_distributed_action); HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::gemm_distributed_action); +HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::trsv_distributed_action); +HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::gemv_distributed_action); +HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::dot_diag_syrk_distributed_action); +HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::dot_diag_gemm_distributed_action); +HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::axpy_distributed_action); diff --git a/examples/distributed/src/distributed_cholesky.hpp b/examples/distributed/src/distributed_cholesky.hpp index 3f4d795a..79207413 100644 --- a/examples/distributed/src/distributed_cholesky.hpp +++ b/examples/distributed/src/distributed_cholesky.hpp @@ -1,154 +1,316 @@ #pragma once +#include "gprat/scheduler.hpp" + #include "distributed_tile.hpp" #include "scheduling.hpp" GPRAT_NS_BEGIN -template -std::vector>> -make_cholesky_dataset(const tiled_scheduler_local &, std::size_t num_tiles) -{ - return { num_tiles * num_tiles }; -} - -// Default implementations in case the scheduler provides none -constexpr std::size_t cholesky_tile(...) { return 0; } - -constexpr std::size_t cholesky_POTRF(...) { return 0; } - -constexpr std::size_t cholesky_SYRK(...) { return 0; } - -constexpr std::size_t cholesky_TRSM(...) { return 0; } - -constexpr std::size_t cholesky_GEMM(...) { return 0; } - -namespace scheduler -{ - -struct tiled_cholesky_scheduler_paap12 : tiled_scheduler_distributed +struct tiled_scheduler_sma : tiled_scheduler_distributed { using tiled_scheduler_distributed::tiled_scheduler_distributed; std::size_t num_localities = localities_.size(); }; -template -tiled_dataset make_cholesky_dataset(const tiled_cholesky_scheduler_paap12 &sched, std::size_t num_tiles) +struct tiled_scheduler_cyclic : tiled_scheduler_distributed { - std::vector> targets; - targets.reserve(sched.num_localities); + using tiled_scheduler_distributed::tiled_scheduler_distributed; - for (std::size_t i = 0; i < sched.num_localities; ++i) + /// @brief Create a new scheduler that targets all localities. + explicit tiled_scheduler_cyclic(std::size_t in_width = 1) : + num_localities(localities_.size()), + width(in_width), + height(num_localities / width) { - targets.emplace_back(sched.localities_[i], 0); + if (num_localities % width != 0) + { + throw std::invalid_argument("num_localities must be divisible by width"); + } } - for (std::size_t row = 0; row < num_tiles; row++) + /// @brief Create a new scheduler that targets the given localities. + explicit tiled_scheduler_cyclic(std::vector in_localities, std::size_t in_width = 1) : + tiled_scheduler_distributed(std::move(in_localities)), + num_localities(localities_.size()), + width(in_width), + height(num_localities / width) { - for (std::size_t col = 0; col < num_tiles; col++) + if (num_localities % width != 0) { - const auto l = (row + col) % sched.num_localities; - ++targets[l].second; + throw std::invalid_argument("num_localities must be divisible by width"); } } - return create_tiled_dataset(targets, num_tiles * num_tiles); + std::size_t num_localities; + std::size_t width; + std::size_t height; +}; + +namespace schedule +{ + +#ifdef _MSC_VER +#pragma warning(push) +#pragma warning(disable : 4100) +#endif + +constexpr std::size_t +covariance_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +{ + return (row + col) % sched.num_localities; } constexpr std::size_t -cholesky_tile(const tiled_cholesky_scheduler_paap12 &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +cross_covariance_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t row, std::size_t col) { return (row + col) % sched.num_localities; } +constexpr std::size_t alpha_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t i) +{ + return (2 * i) % sched.num_localities; +} + +constexpr std::size_t prediction_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t i) +{ + return (2 * i) % sched.num_localities; +} + constexpr std::size_t -cholesky_POTRF(const tiled_cholesky_scheduler_paap12 &sched, std::size_t /*n_tiles*/, std::size_t k) +t_cross_covariance_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t row, std::size_t col) { - return (2 * k) % sched.num_localities; + return (row + col) % sched.num_localities; } constexpr std::size_t -cholesky_SYRK(const tiled_cholesky_scheduler_paap12 &sched, std::size_t /*n_tiles*/, std::size_t m) +prior_K_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t row, std::size_t col) { - return (2 * m) % sched.num_localities; + return (row + col) % sched.num_localities; +} + +constexpr std::size_t +K_inv_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +{ + return (row + col) % sched.num_localities; } constexpr std::size_t -cholesky_TRSM(const tiled_cholesky_scheduler_paap12 &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) +K_grad_v_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +{ + return (row + col) % sched.num_localities; +} + +constexpr std::size_t +K_grad_l_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +{ + return (row + col) % sched.num_localities; +} + +constexpr std::size_t uncertainty_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t i) +{ + return (2 * i) % sched.num_localities; +} + +constexpr std::size_t inter_alpha_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t i) +{ + return (2 * i) % sched.num_localities; +} + +constexpr std::size_t diag_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t i) +{ + return i % sched.num_localities; +} + +constexpr std::size_t cholesky_potrf(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k) +{ + return (2 * k) % sched.num_localities; +} + +constexpr std::size_t cholesky_syrk(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t m) +{ + return (2 * m) % sched.num_localities; +} + +constexpr std::size_t cholesky_trsm(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k, std::size_t m) { return (k + m) % sched.num_localities; } -constexpr std::size_t cholesky_GEMM(const tiled_cholesky_scheduler_paap12 &sched, - std::size_t /*n_tiles*/, - std::size_t /*k*/, - std::size_t m, - std::size_t n) +constexpr std::size_t +cholesky_gemm(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k, std::size_t m, std::size_t n) { return (m + n) % sched.num_localities; } +constexpr std::size_t solve_trsv(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k) +{ + return (2 * k) % sched.num_localities; +} + +constexpr std::size_t solve_trsm(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k) +{ + return (2 * k) % sched.num_localities; +} + +constexpr std::size_t solve_gemv(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k, std::size_t m) +{ + return (k + m) % sched.num_localities; +} + +constexpr std::size_t +solve_matrix_trsm(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t c, std::size_t k) +{ + return (k + c) % sched.num_localities; +} + +constexpr std::size_t +solve_matrix_gemm(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) +{ + return (c + m) % sched.num_localities; +} + +constexpr std::size_t multiply_gemv(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k, std::size_t m) +{ + return (k + m) % sched.num_localities; +} + +constexpr std::size_t k_rank_dot_diag_syrk(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k) +{ + return (2 * k) % sched.num_localities; +} + +constexpr std::size_t +k_rank_gemm(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) +{ + return (k + m) % sched.num_localities; +} + +constexpr std::size_t vector_axpy(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k) +{ + return (2 * k) % sched.num_localities; +} + +constexpr std::size_t get_diagonal(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k) +{ + return (2 * k) % sched.num_localities; +} + +constexpr std::size_t compute_loss(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k) +{ + return (2 * k) % sched.num_localities; +} + // ========================== -struct tiled_cholesky_scheduler_cyclic : tiled_scheduler_distributed +constexpr std::size_t +covariance_tile(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t row, std::size_t col) { - using tiled_scheduler_distributed::tiled_scheduler_distributed; + return (row % sched.height) + (col % sched.width); +} - std::size_t num_localities = localities_.size(); -}; +constexpr std::size_t +cross_covariance_tile(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +{ + return (row % sched.height) + (col % sched.width); +} -template -tiled_dataset make_cholesky_dataset(const tiled_cholesky_scheduler_cyclic &sched, std::size_t num_tiles) +constexpr std::size_t alpha_tile(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t i) { - std::vector> targets; - targets.reserve(sched.num_localities); + return (i % sched.height) + (i % sched.width); +} - for (std::size_t i = 0; i < sched.num_localities; ++i) - { - targets.emplace_back(sched.localities_[i], 0); - } +constexpr std::size_t prediction_tile(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t i) +{ + return (i % sched.height) + (i % sched.width); +} - for (std::size_t row = 0; row < num_tiles; row++) - { - for (std::size_t col = 0; col < num_tiles; col++) - { - const auto l = (row * num_tiles + col) % sched.num_localities; - ++targets[l].second; - } - } +constexpr std::size_t cholesky_potrf(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +{ + return (k % sched.height) + (k % sched.width); +} - return create_tiled_dataset(targets, num_tiles * num_tiles); +constexpr std::size_t cholesky_syrk(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t m) +{ + return (m % sched.height) + (m % sched.width); } constexpr std::size_t -cholesky_tile(const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +cholesky_trsm(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k, std::size_t m) { - return (row * n_tiles + col) % sched.num_localities; + return (m % sched.height) + (k % sched.width); } -constexpr std::size_t cholesky_POTRF(const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +constexpr std::size_t +cholesky_gemm(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k, std::size_t m, std::size_t n) { - return (k * n_tiles + k) % sched.num_localities; + return (m % sched.height) + (n % sched.width); +} + +constexpr std::size_t solve_trsv(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +{ + return (k % sched.height) + (k % sched.width); +} + +constexpr std::size_t solve_trsm(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +{ + return (k % sched.height) + (k % sched.width); +} + +constexpr std::size_t solve_gemv(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k, std::size_t m) +{ + return (k % sched.height) + (m % sched.width); +} + +constexpr std::size_t +solve_matrix_trsm(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t c, std::size_t k) +{ + return (k % sched.height) + (c % sched.width); +} + +constexpr std::size_t +solve_matrix_gemm(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) +{ + return (m % sched.height) + (c % sched.width); +} + +constexpr std::size_t +multiply_gemv(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k, std::size_t m) +{ + return (k % sched.height) + (m % sched.width); } -constexpr std::size_t cholesky_SYRK(const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t m) +constexpr std::size_t k_rank_dot_diag_syrk(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) { - return (m * n_tiles + m) % sched.num_localities; + return (k % sched.height) + (k % sched.width); } constexpr std::size_t -cholesky_TRSM(const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k, std::size_t m) +k_rank_gemm(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) { - return (m * n_tiles + k) % sched.num_localities; + return (k * n_tiles + m) % sched.num_localities; } -constexpr std::size_t cholesky_GEMM( - const tiled_cholesky_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t /*k*/, std::size_t m, std::size_t n) +constexpr std::size_t vector_axpy(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) { - return (m * n_tiles + n) % sched.num_localities; + return (k * n_tiles + k) % sched.num_localities; +} + +constexpr std::size_t get_diagonal(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +{ + return (k * n_tiles + k) % sched.num_localities; } -} // namespace scheduler +constexpr std::size_t compute_loss(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +{ + return (k * n_tiles + k) % sched.num_localities; +} + +#ifdef _MSC_VER +#pragma warning(pop) +#endif + +} // namespace schedule GPRAT_NS_END diff --git a/examples/distributed/src/distributed_tile.cpp b/examples/distributed/src/distributed_tile.cpp index 8807c1d7..5b4e1910 100644 --- a/examples/distributed/src/distributed_tile.cpp +++ b/examples/distributed/src/distributed_tile.cpp @@ -66,7 +66,7 @@ void register_distributed_tile_counters() hpx::performance_counters::install_counter_type( "/gprat/tile_cache/transmission_count", - &get_tile_transmission_time, + &get_tile_transmission_count, "", "", hpx::performance_counters::counter_type::monotonically_increasing); diff --git a/examples/distributed/src/distributed_tile.hpp b/examples/distributed/src/distributed_tile.hpp index db69517a..935341f0 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/examples/distributed/src/distributed_tile.hpp @@ -155,11 +155,11 @@ struct tile_manager_shared_data { } tile_entry(hpx::id_type tile, std::uint32_t locality_id) : - tile(std::move(tile)), + id(std::move(tile)), locality_id(locality_id) { } - hpx::id_type tile; + hpx::id_type id; std::uint32_t locality_id; std::shared_ptr> local_data; @@ -169,7 +169,7 @@ struct tile_manager_shared_data template void serialize(Archive &ar, unsigned) { - ar & tile & locality_id; + ar & id & locality_id; } }; @@ -196,7 +196,7 @@ struct tile_manager : hpx::components::component_base> { if (tile.locality_id == here) { - tile.local_data = hpx::get_ptr>(hpx::launch::sync, tile.tile); + tile.local_data = hpx::get_ptr>(hpx::launch::sync, tile.id); } } } @@ -214,17 +214,17 @@ struct tile_manager : hpx::components::component_base> // Next, try the tile cache - maybe we have current data { mutable_tile_data cached_data; - if (cache_.try_get(target_tile.tile.get_gid(), generation, cached_data)) + if (cache_.try_get(target_tile.id.get_gid(), generation, cached_data)) { return cached_data; } } hpx::chrono::high_resolution_timer timer; - auto data = hpx::async(typename tile_holder::get_data_action{}, target_tile.tile).get(); + auto data = hpx::async(typename tile_holder::get_data_action{}, target_tile.id).get(); record_transmission_time(timer.elapsed_nanoseconds()); - cache_.insert(target_tile.tile.get_gid(), generation, data); + cache_.insert(target_tile.id.get_gid(), generation, data); return data; } @@ -242,20 +242,24 @@ struct tile_manager : hpx::components::component_base> // Next, try the tile cache - maybe we have current data { mutable_tile_data cached_data; - if (cache_.try_get(target_tile.tile.get_gid(), generation, cached_data)) + if (cache_.try_get(target_tile.id.get_gid(), generation, cached_data)) { return hpx::make_ready_future(cached_data); } } - return hpx::async(typename tile_holder::get_data_action{}, target_tile.tile) + return hpx::async(typename tile_holder::get_data_action{}, target_tile.id) .then( - [this, generation, gid = target_tile.tile.get_gid(), timer = hpx::chrono::high_resolution_timer()]( - hpx::future> &&f) + [this, + self = this->get_id(), + generation, + gid = target_tile.id.get_gid(), + timer = hpx::chrono::high_resolution_timer()](hpx::future> &&f) mutable { record_transmission_time(timer.elapsed_nanoseconds()); auto data = f.get(); cache_.insert(gid, generation, data); + self = {}; // release our reference return data; }); } @@ -272,9 +276,9 @@ struct tile_manager : hpx::components::component_base> } // We'd lose this tile after writing it, best to put it in the cache for now - cache_.insert(target_tile.tile.get_gid(), generation, data); + cache_.insert(target_tile.id.get_gid(), generation, data); - return hpx::async(typename tile_holder::set_data_action{}, target_tile.tile, data); + return hpx::async(typename tile_holder::set_data_action{}, target_tile.id, data); } private: @@ -383,7 +387,48 @@ class tile_handle }; template -using tiled_dataset = std::vector>>; +class tiled_dataset +{ + public: + using value_type = hpx::shared_future>; + + tiled_dataset() = default; + + explicit tiled_dataset(std::size_t size) : + data_(std::make_unique(size)), + size_(size) + { } + + [[nodiscard]] std::size_t size() const noexcept { return size_; } + + const value_type *data() const noexcept { return data_.get(); } + + const value_type *begin() const noexcept { return data_.get(); } + + const value_type *end() const noexcept { return data_.get() + size_; } + + value_type &operator[](std::size_t i) + { + if (i >= size_) + { + throw std::out_of_range("tiled_dataset::operator[]"); + } + return data_[i]; + } + + const value_type &operator[](std::size_t i) const + { + if (i >= size_) + { + throw std::out_of_range("tiled_dataset::operator[]"); + } + return data_[i]; + } + + private: + std::unique_ptr data_; + std::size_t size_ = 0; +}; template tiled_dataset @@ -428,11 +473,10 @@ create_tiled_dataset(std::span> targe } // Finally, we create our fat tile_handles - tiled_dataset tiles; - tiles.reserve(num_tiles); + tiled_dataset tiles(num_tiles); for (std::size_t i = 0; i < num_tiles; ++i) { - tiles.push_back(hpx::make_ready_future(tile_handle{ managers, i, 0 })); + tiles[i] = hpx::make_ready_future(tile_handle{ managers, i, 0 }); } return tiles; } diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index d77ff0e5..a66d3736 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -1,32 +1,55 @@ +#include "distributed_blas.hpp" +#include "distributed_cholesky.hpp" +#include "distributed_tile.hpp" + #include "gprat/cpu/gp_algorithms.hpp" +#include "gprat/cpu/gp_functions.hpp" +#include "gprat/gprat.hpp" #include "gprat/kernels.hpp" +#include "gprat/performance_counters.hpp" +#include "gprat/scheduler.hpp" +#include "gprat/utils.hpp" #include "../../test/src/test_data.hpp" -#include "distributed_blas.hpp" -#include "distributed_cholesky.hpp" -#include "distributed_tile.hpp" +#include #include #include #include #include #include +#include #include // This is a standalone test, so including this directly is fine. // Better than having the whole project depend on compiled Boost.Json! - -#include "gprat/gprat.hpp" -#include "gprat/performance_counters.hpp" -#include "gprat/utils.hpp" - #include GPRAT_REGISTER_TILED_DATASET(double, double); GPRAT_NS_BEGIN +template +tiled_dataset make_tiled_dataset(const tiled_scheduler_distributed &sched, std::size_t num_tiles, Mapper &&mapper) +{ + const auto num_localities = sched.localities_.size(); + std::vector> targets; + targets.reserve(num_localities); + + for (std::size_t i = 0; i < num_localities; ++i) + { + targets.emplace_back(sched.localities_[i], 0); + } + + for (std::size_t i = 0; i < num_tiles; i++) + { + ++targets[mapper(i) % num_localities].second; + } + + return create_tiled_dataset(targets, num_tiles); +} + hpx::future> gen_tile_covariance_distributed( - tile_handle tile, + const tile_handle &tile, std::size_t row, std::size_t col, std::size_t N, @@ -40,7 +63,7 @@ GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_covariance, GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_covariance); hpx::future> gen_tile_covariance_distributed( - tile_handle tile, + const tile_handle &tile, std::size_t row, std::size_t col, std::size_t N, @@ -48,107 +71,291 @@ hpx::future> gen_tile_covariance_distributed( const SEKParams &sek_params, const std::vector &input) { - GPRAT_TIME_PLAIN_ACTION(cpu::gen_tile_covariance); return tile.set_async(cpu::gen_tile_covariance(row, col, N, n_regressors, sek_params, input)); } -template -void right_looking_cholesky_tiled(Scheduler &sched, Tiles &tiles, std::size_t N, std::size_t n_tiles) +hpx::future> gen_tile_covariance_with_distance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N, + const SEKParams &sek_params, + const const_tile_data &distance); + +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_covariance_with_distance_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_covariance_with_distance, + gen_tile_covariance_with_distance_distributed_action, + "gen_tile_covariance_with_distance"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_covariance_with_distance); + +hpx::future> gen_tile_covariance_with_distance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N, + const SEKParams &sek_params, + const const_tile_data &distance) { - for (std::size_t k = 0; k < n_tiles; k++) - { - // POTRF: Compute Cholesky factor L - tiles[k * n_tiles + k] = detail::named_dataflow( - sched, cholesky_POTRF(sched, n_tiles, k), "cholesky_tiled", tiles[k * n_tiles + k], N); - for (std::size_t m = k + 1; m < n_tiles; m++) - { - // TRSM: Solve X * L^T = A - tiles[m * n_tiles + k] = detail::named_dataflow( - sched, - cholesky_TRSM(sched, n_tiles, k, m), - "cholesky_tiled", - tiles[k * n_tiles + k], - tiles[m * n_tiles + k], - N, - N, - Blas_trans, - Blas_right); - } - for (std::size_t m = k + 1; m < n_tiles; m++) - { - // SYRK: A = A - B * B^T - tiles[m * n_tiles + m] = detail::named_dataflow( - sched, - cholesky_SYRK(sched, n_tiles, m), - "cholesky_tiled", - tiles[m * n_tiles + m], - tiles[m * n_tiles + k], - N); - for (std::size_t n = k + 1; n < m; n++) - { - // GEMM: C = C - A * B^T - tiles[m * n_tiles + n] = detail::named_dataflow( - sched, - cholesky_GEMM(sched, n_tiles, k, m, n), - "cholesky_tiled", - tiles[m * n_tiles + k], - tiles[n * n_tiles + k], - tiles[m * n_tiles + n], - N, - N, - N, - Blas_no_trans, - Blas_trans); - } - } - } + return tile.set_async(cpu::gen_tile_covariance_with_distance(row, col, N, sek_params, distance)); } -template -std::vector> -cholesky_hpx(Scheduler &sched, - const std::vector &training_input, - const SEKParams &sek_params, - std::size_t n_tiles, - std::size_t n_tile_size, - std::size_t n_regressors) +hpx::future> gen_tile_prior_covariance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &input); +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_prior_covariance_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_prior_covariance, + gen_tile_prior_covariance_distributed_action, + "gen_tile_prior_covariance"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_prior_covariance); + +hpx::future> gen_tile_prior_covariance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &input) { - auto tiles = make_cholesky_dataset(sched, n_tiles); // Tiled covariance matrix + return tile.set_async(cpu::gen_tile_prior_covariance(row, col, N, n_regressors, sek_params, input)); +} - for (std::size_t row = 0; row < n_tiles; row++) - { - for (std::size_t col = 0; col <= row; col++) - { - tiles[row * n_tiles + col] = detail::named_dataflow( - sched, - cholesky_tile(sched, n_tiles, row, col), - "cholesky init", - tiles[row * n_tiles + col], - row, - col, - n_tile_size, - n_regressors, - sek_params, - training_input); - } - } +hpx::future> gen_tile_full_prior_covariance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &input); +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_full_prior_covariance_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_full_prior_covariance, + gen_tile_full_prior_covariance_distributed_action, + "gen_tile_full_prior_covariance"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_prior_covariance); + +hpx::future> gen_tile_full_prior_covariance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &input) +{ + return tile.set_async(cpu::gen_tile_full_prior_covariance(row, col, N, n_regressors, sek_params, input)); +} - /////////////////////////////////////////////////////////////////////////// - // Launch asynchronous Cholesky decomposition: K = L * L^T - right_looking_cholesky_tiled(sched, tiles, n_tile_size, n_tiles); +hpx::future> gen_tile_cross_covariance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N_row, + std::size_t N_col, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &row_input, + const std::vector &col_input); - /////////////////////////////////////////////////////////////////////////// - // Synchronize - std::vector> result(n_tiles * n_tiles); - for (std::size_t i = 0; i < n_tiles; i++) - { - for (std::size_t j = 0; j <= i; j++) - { - result[i * n_tiles + j] = tiles[i * n_tiles + j].get(); - } - } - // hpx::get_runtime_distributed().evaluate_active_counters(false, "POST cholesky"); - return result; +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_cross_covariance_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_cross_covariance, + gen_tile_cross_covariance_distributed_action, + "gen_tile_cross_covariance"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_cross_covariance); + +hpx::future> gen_tile_cross_covariance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N_row, + std::size_t N_col, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &row_input, + const std::vector &col_input) +{ + return tile.set_async( + cpu::gen_tile_cross_covariance(row, col, N_row, N_col, n_regressors, sek_params, row_input, col_input)); +} + +hpx::future> gen_tile_transpose_distributed( + const tile_handle &tile, std::size_t N_row, std::size_t N_col, const tile_handle &src); +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_transpose_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_transpose, gen_tile_transpose_distributed_action, "gen_tile_transpose"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_transpose); + +hpx::future> gen_tile_transpose_distributed( + const tile_handle &tile, std::size_t N_row, std::size_t N_col, const tile_handle &src) +{ + return hpx::dataflow( + hpx::launch::async, + [=](hpx::future> &&tiled) + { return tile.set_async(cpu::gen_tile_transpose(N_row, N_col, tiled.get())); }, + src.get_async()); +} + +hpx::future> gen_tile_output_distributed( + const tile_handle &tile, std::size_t row, std::size_t N, const std::vector &output); + +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_output_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_output, gen_tile_output_distributed_action, "gen_tile_output"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_output); + +hpx::future> gen_tile_output_distributed( + const tile_handle &tile, std::size_t row, std::size_t N, const std::vector &output) +{ + return tile.set_async(cpu::gen_tile_output(row, N, output)); +} + +hpx::future> gen_tile_grad_l_distributed( + const tile_handle &tile, + std::size_t N, + const SEKParams &sek_params, + const const_tile_data &distance); +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_grad_l_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_grad_l, gen_tile_grad_l_distributed_action, "gen_tile_grad_l"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_grad_l); + +hpx::future> gen_tile_grad_l_distributed( + const tile_handle &tile, + std::size_t N, + const SEKParams &sek_params, + const const_tile_data &distance) +{ + return tile.set_async(cpu::gen_tile_grad_l(N, sek_params, distance)); +} + +hpx::future> gen_tile_grad_v_distributed( + const tile_handle &tile, + std::size_t N, + const SEKParams &sek_params, + const const_tile_data &distance); +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_grad_v_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_grad_v, gen_tile_grad_v_distributed_action, "gen_tile_grad_v"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_grad_l); + +hpx::future> gen_tile_grad_v_distributed( + const tile_handle &tile, + std::size_t N, + const SEKParams &sek_params, + const const_tile_data &distance) +{ + return tile.set_async(cpu::gen_tile_grad_v(N, sek_params, distance)); +} + +hpx::future> gen_tile_zeros_distributed(const tile_handle &tile, std::size_t N); + +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_zeros_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_zeros, gen_tile_zeros_distributed_action, "gen_tile_output"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_zeros); + +hpx::future> gen_tile_zeros_distributed(const tile_handle &tile, std::size_t N) +{ + return tile.set_async(cpu::gen_tile_zeros(N)); +} + +hpx::future> gen_tile_identity_distributed(const tile_handle &tile, std::size_t N); + +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_identity_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_identity, gen_tile_identity_distributed_action, "gen_tile_identity"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_identity); + +hpx::future> gen_tile_identity_distributed(const tile_handle &tile, std::size_t N) +{ + return tile.set_async(cpu::gen_tile_identity(N)); +} + +hpx::future> get_matrix_diagonal_distributed(const tile_handle &A, std::size_t M); + +HPX_DEFINE_PLAIN_DIRECT_ACTION(get_matrix_diagonal_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::get_matrix_diagonal, + get_matrix_diagonal_distributed_action, + "get_matrix_diagonal"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::get_matrix_diagonal); + +hpx::future> get_matrix_diagonal_distributed(const tile_handle &A, std::size_t M) +{ + return hpx::dataflow( + hpx::launch::async, + [A, M](hpx::future> &&Ad) + { return A.set_async(cpu::get_matrix_diagonal(Ad.get(), M)); }, + A.get_async()); +} + +hpx::future compute_loss_distributed(const tile_handle &K_diag_tile, + const tile_handle &alpha_tile, + const tile_handle &y_tile, + std::size_t N); + +HPX_DEFINE_PLAIN_DIRECT_ACTION(compute_loss_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::compute_loss, compute_loss_distributed_action, "compute_loss"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::get_matrix_diagonal); + +hpx::future compute_loss_distributed(const tile_handle &K_diag_tile, + const tile_handle &alpha_tile, + const tile_handle &y_tile, + std::size_t N) +{ + return hpx::dataflow( + hpx::launch::async, + [=](hpx::future> &&K_diag_tiled, + hpx::future> &&alpha_tiled, + hpx::future> &&y_tiled) + { return cpu::compute_loss(K_diag_tiled.get(), alpha_tiled.get(), y_tiled.get(), N); }, + K_diag_tile.get_async(), + alpha_tile.get_async(), + y_tile.get_async()); +} + +hpx::future compute_trace_distributed(const tile_handle &diagonal, double trace); + +HPX_DEFINE_PLAIN_DIRECT_ACTION(compute_trace_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::compute_trace, compute_trace_distributed_action, "compute_loss"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::compute_trace); + +hpx::future compute_trace_distributed(const tile_handle &diagonal, double trace) +{ + return hpx::dataflow( + hpx::launch::async, + [=](hpx::future> &&diagonald) { return cpu::compute_trace(diagonald.get(), trace); }, + diagonal.get_async()); +} + +hpx::future +compute_dot_distributed(const tile_handle &vector_T, const tile_handle &vector, double result); + +HPX_DEFINE_PLAIN_DIRECT_ACTION(compute_dot_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::compute_dot, compute_dot_distributed_action, "compute_loss"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::compute_dot); + +hpx::future +compute_dot_distributed(const tile_handle &vector_T, const tile_handle &vector, double result) +{ + return hpx::dataflow( + hpx::launch::async, + [=](hpx::future> &&vector_Td, hpx::future> &&vectord) + { return cpu::compute_dot(vector_Td.get(), vectord.get(), result); }, + vector_T.get_async(), + vector.get_async()); +} + +hpx::future compute_trace_diag_distributed(const tile_handle &tile, double trace, std::size_t N); + +HPX_DEFINE_PLAIN_DIRECT_ACTION(compute_trace_diag_distributed); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::compute_trace_diag, compute_trace_diag_distributed_action, "compute_loss"); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::compute_trace_diag); + +hpx::future compute_trace_diag_distributed(const tile_handle &tile, double trace, std::size_t N) +{ + return hpx::dataflow( + hpx::launch::async, + [=](hpx::future> &&tiled) { return cpu::compute_trace_diag(tiled.get(), trace, N); }, + tile.get_async()); } gprat_results load_test_data_results(const std::string &filename) @@ -204,17 +411,24 @@ void validate_two_dim_result(const std::vector> &expected, } } +void finish_step(const char *name, double elapsed_seconds) +{ + std::cerr << name << " done in " << elapsed_seconds << " seconds" << std::endl; + hpx::evaluate_active_counters(true, name); +} + void run(hpx::program_options::variables_map &vm) { ///////////////////// /////// configuration - std::size_t START = vm["start"].as(); - std::size_t END = vm["end"].as(); - std::size_t STEP = vm["step"].as(); - std::size_t LOOP = vm["loop"].as(); - const int OPT_ITER = vm["opt_iter"].as(); - - int n_test = 1024; + const std::size_t START = vm["start"].as(); + const std::size_t END = vm["end"].as(); + const std::size_t STEP = vm["step"].as(); + const std::size_t LOOP = vm["loop"].as(); + const std::size_t OPT_ITER = vm["opt_iter"].as(); + const std::size_t enabled = vm["enabled"].as(); + + const std::size_t n_test = vm["n_test"].as(); const std::size_t n_tiles = vm["tiles"].as(); const std::size_t n_reg = vm["regressors"].as(); @@ -223,29 +437,30 @@ void run(hpx::program_options::variables_map &vm) const auto &test_path = vm["test_path"].as(); std::optional test_results; - // XXX: cannot use contains() because it's not exported by HPX program_options - // ReSharper disable once CppUseAssociativeContains - if (vm.find("test_results_path") != vm.end()) + const auto test_results_path = vm["test_results_path"].as(); + if (!test_results_path.empty()) { - test_results = load_test_data_results(vm["test_results_path"].as()); + test_results = load_test_data_results(test_results_path); std::cerr << "We have comparison data!" << std::endl; } - scheduler::tiled_cholesky_scheduler_paap12 scheduler; + tiled_scheduler_sma scheduler; for (std::size_t start = START; start <= END; start = start * STEP) { - int n_train = static_cast(start); + const auto n_train = start; for (std::size_t l = 0; l < LOOP; l++) { hpx::chrono::high_resolution_timer total_timer; // Compute tile sizes and number of predict tiles - int tile_size = compute_train_tile_size(n_train, n_tiles); - auto result = compute_test_tiles(n_test, n_tiles, tile_size); + const auto tile_size = compute_train_tile_size(n_train, n_tiles); + const auto result = compute_test_tiles(n_test, n_tiles, tile_size); ///////////////////// ///// hyperparams AdamParams hpar = { 0.1, 0.9, 0.999, 1e-8, OPT_ITER }; + SEKParams sek_params = { 1.0, 1.0, 0.1 }; + std::vector trainable = { true, true, true }; ///////////////////// ////// data loading @@ -255,20 +470,92 @@ void run(hpx::program_options::variables_map &vm) ///////////////////// ///// GP - hpx::chrono::high_resolution_timer init_timer; - std::vector trainable = { true, true, true }; - GP gp(training_input.data, training_output.data, n_tiles, tile_size, n_reg, { 1.0, 1.0, 0.1 }, trainable); - const auto init_time = init_timer.elapsed(); + gprat_results results; - // Measure the time taken to execute gp.cholesky(); - auto start_cholesky = std::chrono::high_resolution_clock::now(); + // Start with a clean slate + hpx::reset_active_counters(); hpx::chrono::high_resolution_timer cholesky_timer; - const auto cholesky = - cholesky_hpx(scheduler, training_input.data, { 1.0, 1.0, 0.1 }, n_tiles, tile_size, n_reg); + if (enabled & (1 << 0)) + { + results.choleksy = + to_vector(cpu::cholesky(scheduler, training_input.data, sek_params, n_tiles, tile_size, n_reg)); + } const auto cholesky_time = cholesky_timer.elapsed(); + finish_step("cholesky", cholesky_time); + + hpx::chrono::high_resolution_timer opt_timer; + if (enabled & (1 << 1)) + { + results.losses = cpu::optimize( + scheduler, + training_input.data, + training_output.data, + n_tiles, + tile_size, + n_reg, + hpar, + sek_params, + trainable); + } + const auto opt_time = opt_timer.elapsed(); + finish_step("opt", opt_time); - // Save parameters and times to a .txt file with a header + hpx::chrono::high_resolution_timer predict_timer; + if (enabled & (1 << 2)) + { + results.pred = cpu::predict( + scheduler, + training_input.data, + training_output.data, + test_input.data, + sek_params, + n_tiles, + tile_size, + result.first, + result.second, + n_reg); + } + const auto predict_time = predict_timer.elapsed(); + finish_step("predict", predict_time); + + hpx::chrono::high_resolution_timer predict_with_uncertainty_timer; + if (enabled & (1 << 3)) + { + results.sum = cpu::predict_with_uncertainty( + scheduler, + training_input.data, + training_output.data, + test_input.data, + sek_params, + n_tiles, + tile_size, + result.first, + result.second, + n_reg); + } + const auto predict_with_uncertainty_time = predict_with_uncertainty_timer.elapsed(); + finish_step("predict_with_uncertainty", predict_with_uncertainty_time); + + hpx::chrono::high_resolution_timer predict_with_full_cov_timer; + if (enabled & (1 << 4)) + { + results.full = cpu::predict_with_full_cov( + scheduler, + training_input.data, + training_output.data, + test_input.data, + sek_params, + n_tiles, + tile_size, + result.first, + result.second, + n_reg); + } + const auto predict_with_full_cov_time = predict_with_full_cov_timer.elapsed(); + finish_step("predict_with_full_cov", predict_with_full_cov_time); + + // Save parameters and times to a .csv file with a header std::ofstream outfile(vm["timings_csv"].as(), std::ios::app); if (outfile.tellp() == 0) { @@ -277,15 +564,70 @@ void run(hpx::program_options::variables_map &vm) "Opt_time,Pred_Uncer_time,Pred_Full_time,Pred_time,N_loop\n"; } outfile << hpx::get_locality_id() << "," << n_train << "," << n_test << "," << n_tiles << "," << n_reg - << "," << OPT_ITER << "," << total_timer.elapsed() << "," << init_time << "," << cholesky_time - << "," << 0 << "," << 0 << "," << 0 << "," << 0 << "," << l << "\n"; + << "," << OPT_ITER << "," << total_timer.elapsed() << "," << 0 << "," << cholesky_time << "," + << opt_time << "," << predict_with_uncertainty_time << "," << predict_with_full_cov_time << "," + << predict_time << "," << l << "\n"; outfile.close(); if (test_results) { +#define REQUIRE(expr) \ + if (!expr) \ + throw std::runtime_error(#expr); +#define REQUIRE_THAT(a, b) \ + if (!b.match(a)) \ + throw std::runtime_error(std::format("{} != {}: {} {}", #a, #b, a, b.describe())); + const auto &expected_results = *test_results; std::cerr << "Validating results..." << std::endl; - validate_two_dim_result(test_results->choleksy, cholesky); + REQUIRE(results.choleksy.size() == expected_results.choleksy.size()); + REQUIRE(results.losses.size() == expected_results.losses.size()); + REQUIRE(results.sum.size() == expected_results.sum.size()); + REQUIRE(results.sum[0].size() == expected_results.sum[0].size()); + REQUIRE(results.full.size() == expected_results.full.size()); + REQUIRE(results.full[0].size() == expected_results.full[0].size()); + REQUIRE(results.pred.size() == expected_results.pred.size()); + + // Now we can compare content + // The default-constructed WithinRel() matcher has a tolerance of epsilon * 100 + // see: + // https://github.com/catchorg/Catch2/blob/914aeecfe23b1e16af6ea675a4fb5dbd5a5b8d0a/docs/comparing-floating-point-numbers.md#withinrel + using Catch::Matchers::WithinRel; + double eps = std::numeric_limits::epsilon() * 1'000'000; + for (std::size_t i = 0, n = results.choleksy.size(); i != n; ++i) + { + for (std::size_t j = 0, m = results.choleksy[i].size(); j != m; ++j) + { + REQUIRE_THAT(results.choleksy[i][j], WithinRel(expected_results.choleksy[i][j], eps)); + } + } + for (std::size_t i = 0, n = results.losses.size(); i != n; ++i) + { + REQUIRE_THAT(results.losses[i], WithinRel(expected_results.losses[i], eps)); + } + + for (std::size_t i = 0, n = results.full.size(); i != n; ++i) + { + for (std::size_t j = 0, m = results.full[i].size(); j != m; ++j) + { + REQUIRE_THAT(results.full[i][j], WithinRel(expected_results.full[i][j], eps)); + } + } + + for (std::size_t i = 0, n = results.sum.size(); i != n; ++i) + { + for (std::size_t j = 0, m = results.sum[i].size(); j != m; ++j) + { + REQUIRE_THAT(results.sum[i][j], WithinRel(expected_results.sum[i][j], eps)); + } + } + + for (std::size_t i = 0, n = results.pred.size(); i != n; ++i) + { + REQUIRE_THAT(results.pred[i], WithinRel(expected_results.pred[i], eps)); + } } + + std::cerr << "====================" << std::endl; } } std::cerr << "DONE!" << std::endl; @@ -301,7 +643,6 @@ void startup() { register_performance_counters(); register_distributed_tile_counters(); - register_distributed_blas_counters(); } } once_dummy; } @@ -366,8 +707,10 @@ int main(int argc, char *argv[]) ("start", po::value()->default_value(128), "Starting number of training samples") ("end", po::value()->default_value(128), "End number of training samples") ("step", po::value()->default_value(2), "Increment of training samples") + ("n_test", po::value()->default_value(128), "Number of test samples") ("loop", po::value()->default_value(1), "Number of iterations to be performed for each number of training samples") - ("opt_iter", po::value()->default_value(3), "Number of optimization iterations*/") + ("opt_iter", po::value()->default_value(3), "Number of optimization iterations*/") + ("enabled", po::value()->default_value((std::numeric_limits::max)()), "Bitmask of enabled steps") ; // clang-format on diff --git a/examples/distributed/src/scheduling.cpp b/examples/distributed/src/scheduling.cpp index 3917124d..6e1d0489 100644 --- a/examples/distributed/src/scheduling.cpp +++ b/examples/distributed/src/scheduling.cpp @@ -1,67 +1,2 @@ #include "scheduling.hpp" - -#include -#include - -std::atomic tile_transmission_time(0); - -void record_transmission_time(std::int64_t elapsed_ns) -{ - HPX_ASSERT(elapsed_ns >= 0); - tile_transmission_time += elapsed_ns; -} - -std::uint64_t get_transmission_time(bool reset) -{ - return hpx::util::get_and_reset_value(tile_transmission_time, reset); -} - -void register_distributed_tile_counters() -{ - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/hits", - &tile_cache_counters::get_cache_hits, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/misses", - &tile_cache_counters::get_cache_misses, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/evictions", - &tile_cache_counters::get_cache_evictions, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/insertions", - &tile_cache_counters::get_cache_insertions, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/transmission_time", - &get_transmission_time, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); -} - -// The macros below are necessary to generate the code required for exposing -// our partition type remotely. -// -// HPX_REGISTER_COMPONENT() exposes the component creation -// through hpx::new_<>(). -typedef hpx::components::component tile_server_type; -HPX_REGISTER_COMPONENT(tile_server_type, tile_server) - -// HPX_REGISTER_ACTION() exposes the component member function for remote -// invocation. -typedef tile_server::get_data_action get_data_action; -HPX_REGISTER_ACTION(get_data_action) - -typedef tile_server::set_data_action set_data_action; -HPX_REGISTER_ACTION(set_data_action) +#include "distributed_tile.hpp" diff --git a/examples/distributed/src/scheduling.hpp b/examples/distributed/src/scheduling.hpp index 919a251b..a735b06d 100644 --- a/examples/distributed/src/scheduling.hpp +++ b/examples/distributed/src/scheduling.hpp @@ -2,126 +2,9 @@ #include "gprat/detail/config.hpp" -#include -#include -#include -#include -#include +#include "gprat/detail/async_helpers.hpp" +#include "gprat/detail/actions.hpp" GPRAT_NS_BEGIN -template -struct plain_action_for; - -#define GPRAT_DECLARE_PLAIN_ACTION_FOR(local_function, action, friendly_name) \ - template <> \ - struct plain_action_for \ - { \ - using action_type = action; \ - constexpr static std::string_view name = friendly_name; \ - static std::atomic elapsed_ns_in_action; \ - } - -#define GPRAT_DEFINE_PLAIN_ACTION_FOR(local_function) \ - std::atomic plain_action_for::elapsed_ns_in_action(0) - -struct plain_action_timer -{ - explicit plain_action_timer(std::atomic &total) : - total(total) - { } - - ~plain_action_timer() - { - const auto elapsed = timer.elapsed_nanoseconds(); - HPX_ASSERT(elapsed >= 0); - if (elapsed > 0) - total += static_cast(elapsed); - } - - std::atomic &total; - hpx::chrono::high_resolution_timer timer; -}; - -#define GPRAT_TIME_PLAIN_ACTION(local_function) \ - plain_action_timer _action_timer(plain_action_for::elapsed_ns_in_action); - -template -std::uint64_t get_and_reset_plain_action_elapsed(bool reset) -{ - return hpx::util::get_and_reset_value(plain_action_for::elapsed_ns_in_action, reset); -} - -struct tiled_scheduler_distributed -{ - tiled_scheduler_distributed() : - localities_(hpx::find_all_localities()) - { - // ctor - } - - explicit tiled_scheduler_distributed(std::vector in_localities) : - localities_(std::move(in_localities)) - { - // ctor - } - - std::vector localities_; -}; - -struct tiled_scheduler_local -{ }; - -namespace detail -{ -// HPX does not auto-collapse future chains in their async(), dataflow(), ... functions. -// This usually works fine, but we require shared_futures most of the time -// and the language will not do two-step conversions for us (future> -> future -> shared_future). -// see: https://github.com/STEllAR-GROUP/hpx/issues/3758 -template -hpx::future collapse(hpx::future> &&fut) -{ - return { std::move(fut) }; -} - -template -hpx::future collapse(hpx::future &&fut) -{ - return std::move(fut); -} - -template -decltype(auto) -named_dataflow(const tiled_scheduler_distributed &sched, std::size_t on, const char * /*name*/, Args &&...args) -{ - return collapse(hpx::dataflow(hpx::launch::async, - hpx::unwrapping(typename plain_action_for::action_type{}), - sched.localities_[on], - std::forward(args)...)); - /*return hpx::dataflow( - [timer = hpx::chrono::high_resolution_timer()](auto &&r) - { - const auto elapsed = timer.elapsed_nanoseconds(); - HPX_ASSERT(elapsed >= 0); - plain_action_for::elapsed_ns_in_action += elapsed; - return collapse(r.get()); - }, - hpx::dataflow(typename plain_action_for::action_type{}, sched.localities_[on], - std::forward(args)...));*/ -} - -template -decltype(auto) named_dataflow(const tiled_scheduler_local &sched, std::size_t /*on*/, const char *name, Args &&...args) -{ - return hpx::dataflow(hpx::annotated_function(hpx::unwrapping(F), name), std::forward(args)...); -} - -template -decltype(auto) named_async(const tiled_scheduler_local &sched, std::size_t /*on*/, const char *name, Args &&...args) -{ - return hpx::async(hpx::annotated_function(F, name), std::forward(args)...); -} - -} // namespace detail - GPRAT_NS_END diff --git a/test/src/output_correctness.cpp b/test/src/output_correctness.cpp index 89dee4af..52aa5569 100644 --- a/test/src/output_correctness.cpp +++ b/test/src/output_correctness.cpp @@ -14,36 +14,6 @@ #include #include -template -std::vector to_vector(const gprat::const_tile_data &data) -{ - return { data.begin(), data.end() }; -} - -template -std::vector> to_vector(const std::vector> &data) -{ - std::vector> out; - out.reserve(data.size()); - for (const auto &row : data) - { - out.emplace_back(to_vector(row)); - } - return out; -} - -template -std::vector> to_vector(const std::vector> &data) -{ - std::vector> out; - out.reserve(data.size()); - for (const auto &row : data) - { - out.emplace_back(to_vector(row)); - } - return out; -} - // This logic is basically equivalent to the GPRat C++ example (for now). gprat_results run_on_data_cpu(const std::string &train_path, const std::string &out_path, const std::string &test_path) { diff --git a/test/src/test_data.hpp b/test/src/test_data.hpp index aa759446..b9d31b62 100644 --- a/test/src/test_data.hpp +++ b/test/src/test_data.hpp @@ -1,5 +1,7 @@ #pragma once +#include "gprat/gprat.hpp" + #include #include @@ -52,3 +54,33 @@ inline gprat_results tag_invoke(boost::json::value_to_tag, const extract(obj, results.pred_no_optimize, "pred_no_optimize"); return results; } + +template +std::vector to_vector(const gprat::const_tile_data &data) +{ + return { data.begin(), data.end() }; +} + +template +std::vector> to_vector(const std::vector> &data) +{ + std::vector> out; + out.reserve(data.size()); + for (const auto &row : data) + { + out.emplace_back(to_vector(row)); + } + return out; +} + +template +std::vector> to_vector(const std::vector> &data) +{ + std::vector> out; + out.reserve(data.size()); + for (const auto &row : data) + { + out.emplace_back(to_vector(row)); + } + return out; +} From cd77120bccb3cb0a37b475ed3bd2a5ff4d06ebb3 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Sun, 30 Nov 2025 23:07:58 +0100 Subject: [PATCH 55/56] refactor!(core): Properly move distributed code from examples to core --- core/CMakeLists.txt | 11 + .../gprat/cpu/adapter_cblas_fp64_actions.hpp | 55 +-- .../gprat/cpu/gp_algorithms_actions.hpp | 104 +++++ .../gprat/cpu/gp_optimizer_actions.hpp | 80 ++++ .../gprat/cpu/gp_uncertainty_actions.hpp | 25 ++ core/include/gprat/detail/actions.hpp | 6 +- core/include/gprat/performance_counters.hpp | 5 + core/include/gprat/scheduler/cyclic.hpp | 157 ++++++++ core/include/gprat/scheduler/sma.hpp | 175 +++++++++ core/include/gprat/tile_cache.hpp | 97 +++++ core/include/gprat/tile_data.hpp | 11 +- .../include/gprat/tiled_dataset.hpp | 127 ++----- .../src/cpu/adapter_cblas_fp64_actions.cpp | 35 +- core/src/cpu/gp_algorithms_actions.cpp | 100 +++++ core/src/cpu/gp_optimizer_actions.cpp | 93 +++++ core/src/cpu/gp_uncertainty_actions.cpp | 23 ++ core/src/gprat.cpp | 1 + core/src/performance_counters.cpp | 40 ++ core/src/tile_cache.cpp | 15 + core/src/tiled_dataset.cpp | 3 + examples/distributed/CMakeLists.txt | 13 +- .../distributed/src/distributed_cholesky.hpp | 316 ---------------- examples/distributed/src/distributed_tile.cpp | 95 ----- examples/distributed/src/main.cpp | 355 +----------------- examples/distributed/src/scheduling.cpp | 2 - examples/distributed/src/scheduling.hpp | 10 - 26 files changed, 1034 insertions(+), 920 deletions(-) rename examples/distributed/src/distributed_blas.hpp => core/include/gprat/cpu/adapter_cblas_fp64_actions.hpp (50%) create mode 100644 core/include/gprat/cpu/gp_algorithms_actions.hpp create mode 100644 core/include/gprat/cpu/gp_optimizer_actions.hpp create mode 100644 core/include/gprat/cpu/gp_uncertainty_actions.hpp create mode 100644 core/include/gprat/scheduler/cyclic.hpp create mode 100644 core/include/gprat/scheduler/sma.hpp create mode 100644 core/include/gprat/tile_cache.hpp rename examples/distributed/src/distributed_tile.hpp => core/include/gprat/tiled_dataset.hpp (81%) rename examples/distributed/src/distributed_blas.cpp => core/src/cpu/adapter_cblas_fp64_actions.cpp (80%) create mode 100644 core/src/cpu/gp_algorithms_actions.cpp create mode 100644 core/src/cpu/gp_optimizer_actions.cpp create mode 100644 core/src/cpu/gp_uncertainty_actions.cpp create mode 100644 core/src/tile_cache.cpp create mode 100644 core/src/tiled_dataset.cpp delete mode 100644 examples/distributed/src/distributed_cholesky.hpp delete mode 100644 examples/distributed/src/distributed_tile.cpp delete mode 100644 examples/distributed/src/scheduling.cpp delete mode 100644 examples/distributed/src/scheduling.hpp diff --git a/core/CMakeLists.txt b/core/CMakeLists.txt index 1a7b4db3..58a65d10 100644 --- a/core/CMakeLists.txt +++ b/core/CMakeLists.txt @@ -23,6 +23,17 @@ set(SOURCE_FILES src/cpu/adapter_cblas_fp32.cpp src/cpu/adapter_cblas_fp64.cpp) +# GPRat distributed TODO: this could be gated behind a distributed-only option! +list( + APPEND + SOURCE_FILES + src/cpu/adapter_cblas_fp64_actions.cpp + src/cpu/gp_algorithms_actions.cpp + src/cpu/gp_uncertainty_actions.cpp + src/cpu/gp_optimizer_actions.cpp + src/tile_cache.cpp + src/tiled_dataset.cpp) + if(GPRAT_WITH_CUDA) list( APPEND diff --git a/examples/distributed/src/distributed_blas.hpp b/core/include/gprat/cpu/adapter_cblas_fp64_actions.hpp similarity index 50% rename from examples/distributed/src/distributed_blas.hpp rename to core/include/gprat/cpu/adapter_cblas_fp64_actions.hpp index 7b19283b..442bd929 100644 --- a/examples/distributed/src/distributed_blas.hpp +++ b/core/include/gprat/cpu/adapter_cblas_fp64_actions.hpp @@ -1,15 +1,20 @@ -#pragma once +#ifndef GPRAT_CPU_ADAPTER_CBLAS_FP64_ACTIONS_HPP +#define GPRAT_CPU_ADAPTER_CBLAS_FP64_ACTIONS_HPP + +#pragma once #include "gprat/cpu/adapter_cblas_fp64.hpp" +#include "gprat/detail/actions.hpp" +#include "gprat/detail/config.hpp" +#include "gprat/tiled_dataset.hpp" -#include "distributed_tile.hpp" -#include "scheduling.hpp" #include -GPRAT_REGISTER_TILED_DATASET_DECLARATION(double, double); - GPRAT_NS_BEGIN +namespace cpu +{ + hpx::future> potrf_distributed(const tile_handle &A, int N); hpx::future> trsm_distributed( const tile_handle &L, @@ -43,10 +48,10 @@ hpx::future> gemv_distributed( hpx::future> dot_diag_syrk_distributed(const tile_handle &A, const tile_handle &r, int N, int M); hpx::future> dot_diag_gemm_distributed( -const tile_handle &A, const tile_handle &B, const tile_handle &r, int N, int M); -hpx::future> axpy_distributed( - const tile_handle &y, const tile_handle &x, int N); + const tile_handle &A, const tile_handle &B, const tile_handle &r, int N, int M); +hpx::future> axpy_distributed(const tile_handle &y, const tile_handle &x, int N); +// This just gives us the action type (that we want in the correct namespace) HPX_DEFINE_PLAIN_DIRECT_ACTION(potrf_distributed); HPX_DEFINE_PLAIN_DIRECT_ACTION(trsm_distributed); HPX_DEFINE_PLAIN_DIRECT_ACTION(syrk_distributed); @@ -57,24 +62,22 @@ HPX_DEFINE_PLAIN_DIRECT_ACTION(dot_diag_syrk_distributed); HPX_DEFINE_PLAIN_DIRECT_ACTION(dot_diag_gemm_distributed); HPX_DEFINE_PLAIN_DIRECT_ACTION(axpy_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&potrf, potrf_distributed_action, "POTRF"); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&trsm, trsm_distributed_action, "TRSM"); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&syrk, syrk_distributed_action, "SYRK"); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&gemm, gemm_distributed_action, "GEMM"); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&trsv, trsv_distributed_action, "TRSV"); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&gemv, gemv_distributed_action, "GEMV"); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&dot_diag_syrk, dot_diag_syrk_distributed_action, "dot diag(SYRK)"); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&dot_diag_gemm, dot_diag_gemm_distributed_action, "dot diag(GEMM)"); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&axpy, axpy_distributed_action, "axpy"); +} // namespace cpu GPRAT_NS_END -HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::potrf_distributed_action); -HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::trsm_distributed_action); -HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::syrk_distributed_action); -HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::gemm_distributed_action); -HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::trsv_distributed_action); -HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::gemv_distributed_action); -HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::dot_diag_syrk_distributed_action); -HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::dot_diag_gemm_distributed_action); -HPX_REGISTER_ACTION_DECLARATION(GPRAT_NS::axpy_distributed_action); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::potrf, GPRAT_NS::cpu::potrf_distributed_action, "POTRF"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::trsm, GPRAT_NS::cpu::trsm_distributed_action, "TRSM"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::syrk, GPRAT_NS::cpu::syrk_distributed_action, "SYRK"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::gemm, GPRAT_NS::cpu::gemm_distributed_action, "GEMM"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::trsv, GPRAT_NS::cpu::trsv_distributed_action, "TRSV"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::gemv, GPRAT_NS::cpu::gemv_distributed_action, "GEMV"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::dot_diag_syrk, + GPRAT_NS::cpu::dot_diag_syrk_distributed_action, + "dot diag(SYRK)"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::dot_diag_gemm, + GPRAT_NS::cpu::dot_diag_gemm_distributed_action, + "dot diag(GEMM)"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::axpy, GPRAT_NS::cpu::axpy_distributed_action, "axpy"); + +#endif diff --git a/core/include/gprat/cpu/gp_algorithms_actions.hpp b/core/include/gprat/cpu/gp_algorithms_actions.hpp new file mode 100644 index 00000000..578cad37 --- /dev/null +++ b/core/include/gprat/cpu/gp_algorithms_actions.hpp @@ -0,0 +1,104 @@ +#ifndef GPRAT_CPU_GP_ALGORITHMS_ACTIONS_HPP +#define GPRAT_CPU_GP_ALGORITHMS_ACTIONS_HPP + +#pragma once + +#include "gprat/cpu/gp_algorithms.hpp" +#include "gprat/detail/actions.hpp" +#include "gprat/detail/config.hpp" +#include "gprat/tiled_dataset.hpp" + +GPRAT_NS_BEGIN + +namespace cpu +{ + +hpx::future> gen_tile_covariance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &input); +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_covariance_distributed); + +hpx::future> gen_tile_prior_covariance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &input); +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_prior_covariance_distributed); + +hpx::future> gen_tile_full_prior_covariance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &input); +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_full_prior_covariance_distributed); + +hpx::future> gen_tile_cross_covariance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N_row, + std::size_t N_col, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &row_input, + const std::vector &col_input); + +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_cross_covariance_distributed); + +hpx::future> gen_tile_transpose_distributed( + const tile_handle &tile, std::size_t N_row, std::size_t N_col, const tile_handle &src); +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_transpose_distributed); + +hpx::future> gen_tile_output_distributed( + const tile_handle &tile, std::size_t row, std::size_t N, const std::vector &output); + +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_output_distributed); + +hpx::future> gen_tile_zeros_distributed(const tile_handle &tile, std::size_t N); + +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_zeros_distributed); + +hpx::future> gen_tile_identity_distributed(const tile_handle &tile, std::size_t N); + +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_identity_distributed); +} // namespace cpu + +GPRAT_NS_END + +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_covariance, + GPRAT_NS::cpu::gen_tile_covariance_distributed_action, + "cpu::gen_tile_covariance"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_prior_covariance, + GPRAT_NS::cpu::gen_tile_prior_covariance_distributed_action, + "gen_tile_prior_covariance"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_full_prior_covariance, + GPRAT_NS::cpu::gen_tile_full_prior_covariance_distributed_action, + "gen_tile_full_prior_covariance"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_cross_covariance, + GPRAT_NS::cpu::gen_tile_cross_covariance_distributed_action, + "gen_tile_cross_covariance"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_transpose, + GPRAT_NS::cpu::gen_tile_transpose_distributed_action, + "gen_tile_transpose"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_output, + GPRAT_NS::cpu::gen_tile_output_distributed_action, + "gen_tile_output"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_zeros, + GPRAT_NS::cpu::gen_tile_zeros_distributed_action, + "gen_tile_output"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_identity, + GPRAT_NS::cpu::gen_tile_identity_distributed_action, + "gen_tile_identity"); + +#endif diff --git a/core/include/gprat/cpu/gp_optimizer_actions.hpp b/core/include/gprat/cpu/gp_optimizer_actions.hpp new file mode 100644 index 00000000..ade9de4e --- /dev/null +++ b/core/include/gprat/cpu/gp_optimizer_actions.hpp @@ -0,0 +1,80 @@ +#ifndef GPRAT_CPU_GP_OPTIMIZER_ACTIONS_HPP +#define GPRAT_CPU_GP_OPTIMIZER_ACTIONS_HPP + +#pragma once + +#include "gprat/cpu/gp_optimizer.hpp" +#include "gprat/detail/actions.hpp" +#include "gprat/detail/config.hpp" +#include "gprat/tiled_dataset.hpp" + +GPRAT_NS_BEGIN + +namespace cpu +{ +hpx::future> gen_tile_covariance_with_distance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N, + const SEKParams &sek_params, + const const_tile_data &distance); +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_covariance_with_distance_distributed); + +hpx::future> gen_tile_grad_l_distributed( + const tile_handle &tile, + std::size_t N, + const SEKParams &sek_params, + const const_tile_data &distance); +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_grad_l_distributed); + +hpx::future> gen_tile_grad_v_distributed( + const tile_handle &tile, + std::size_t N, + const SEKParams &sek_params, + const const_tile_data &distance); +HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_grad_v_distributed); + +hpx::future compute_loss_distributed(const tile_handle &K_diag_tile, + const tile_handle &alpha_tile, + const tile_handle &y_tile, + std::size_t N); +HPX_DEFINE_PLAIN_DIRECT_ACTION(compute_loss_distributed); + +hpx::future compute_trace_distributed(const tile_handle &diagonal, double trace); +HPX_DEFINE_PLAIN_DIRECT_ACTION(compute_trace_distributed); + +hpx::future +compute_dot_distributed(const tile_handle &vector_T, const tile_handle &vector, double result); +HPX_DEFINE_PLAIN_DIRECT_ACTION(compute_dot_distributed); + +hpx::future compute_trace_diag_distributed(const tile_handle &tile, double trace, std::size_t N); +HPX_DEFINE_PLAIN_DIRECT_ACTION(compute_trace_diag_distributed); + +} // namespace cpu + +GPRAT_NS_END + +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_covariance_with_distance, + GPRAT_NS::cpu::gen_tile_covariance_with_distance_distributed_action, + "gen_tile_covariance_with_distance"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_grad_l, + GPRAT_NS::cpu::gen_tile_grad_l_distributed_action, + "gen_tile_grad_l"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_grad_v, + GPRAT_NS::cpu::gen_tile_grad_v_distributed_action, + "gen_tile_grad_v"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::compute_loss, + GPRAT_NS::cpu::compute_loss_distributed_action, + "compute_loss"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::compute_trace, + GPRAT_NS::cpu::compute_trace_distributed_action, + "compute_trace"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::compute_dot, + GPRAT_NS::cpu::compute_dot_distributed_action, + "compute_dot"); +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::compute_trace_diag, + GPRAT_NS::cpu::compute_trace_diag_distributed_action, + "compute_trace_diag"); + +#endif diff --git a/core/include/gprat/cpu/gp_uncertainty_actions.hpp b/core/include/gprat/cpu/gp_uncertainty_actions.hpp new file mode 100644 index 00000000..31e47d88 --- /dev/null +++ b/core/include/gprat/cpu/gp_uncertainty_actions.hpp @@ -0,0 +1,25 @@ +#ifndef GPRAT_CPU_GP_UNCERTAINTY_ACTIONS_HPP +#define GPRAT_CPU_GP_UNCERTAINTY_ACTIONS_HPP + +#pragma once + +#include "gprat/cpu/gp_uncertainty.hpp" +#include "gprat/detail/actions.hpp" +#include "gprat/detail/config.hpp" +#include "gprat/tiled_dataset.hpp" + +GPRAT_NS_BEGIN + +namespace cpu +{ +hpx::future> get_matrix_diagonal_distributed(const tile_handle &A, std::size_t M); +HPX_DEFINE_PLAIN_DIRECT_ACTION(get_matrix_diagonal_distributed); +} // namespace cpu + +GPRAT_NS_END + +GPRAT_DECLARE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::get_matrix_diagonal, + GPRAT_NS::cpu::get_matrix_diagonal_distributed_action, + "get_matrix_diagonal"); + +#endif diff --git a/core/include/gprat/detail/actions.hpp b/core/include/gprat/detail/actions.hpp index cc047fbb..6d6cfb21 100644 --- a/core/include/gprat/detail/actions.hpp +++ b/core/include/gprat/detail/actions.hpp @@ -7,6 +7,7 @@ #include #include +#include #include GPRAT_NS_BEGIN @@ -20,14 +21,15 @@ template struct plain_action_for; #define GPRAT_DECLARE_PLAIN_ACTION_FOR(local_function, action, friendly_name) \ + HPX_REGISTER_ACTION_DECLARATION(action) \ template <> \ - struct plain_action_for \ + struct GPRAT_NS::plain_action_for \ { \ using action_type = action; \ constexpr static std::string_view name = friendly_name; \ } -#define GPRAT_DEFINE_PLAIN_ACTION_FOR(local_function) +#define GPRAT_DEFINE_PLAIN_ACTION_FOR(local_function, action) HPX_REGISTER_ACTION(action) // ============================================================= // distributed action-based scheduling diff --git a/core/include/gprat/performance_counters.hpp b/core/include/gprat/performance_counters.hpp index 13054735..04cdf63c 100644 --- a/core/include/gprat/performance_counters.hpp +++ b/core/include/gprat/performance_counters.hpp @@ -77,6 +77,11 @@ std::uint64_t get_and_reset_function_calls(bool reset) void track_tile_data_allocation(std::size_t size); void track_tile_data_deallocation(std::size_t size); +void track_tile_server_allocation(std::size_t size); +void track_tile_server_deallocation(std::size_t size); + +void record_transmission_time(std::int64_t elapsed_ns); + void register_performance_counters(); void force_evict_memory(const void *start, std::size_t size); diff --git a/core/include/gprat/scheduler/cyclic.hpp b/core/include/gprat/scheduler/cyclic.hpp new file mode 100644 index 00000000..71f58d77 --- /dev/null +++ b/core/include/gprat/scheduler/cyclic.hpp @@ -0,0 +1,157 @@ +#ifndef GPRAT_SCHEDULER_CYCLIC_HPP +#define GPRAT_SCHEDULER_CYCLIC_HPP + +#pragma once + +#include "gprat/detail/actions.hpp" +#include "gprat/detail/config.hpp" +#include "gprat/scheduler.hpp" + +GPRAT_NS_BEGIN + +struct tiled_scheduler_cyclic : tiled_scheduler_distributed +{ + using tiled_scheduler_distributed::tiled_scheduler_distributed; + + /// @brief Create a new scheduler that targets all localities. + explicit tiled_scheduler_cyclic(std::size_t in_width = 1) : + num_localities(localities_.size()), + width(in_width), + height(num_localities / width) + { + if (num_localities % width != 0) + { + throw std::invalid_argument("num_localities must be divisible by width"); + } + } + + /// @brief Create a new scheduler that targets the given localities. + explicit tiled_scheduler_cyclic(std::vector in_localities, std::size_t in_width = 1) : + tiled_scheduler_distributed(std::move(in_localities)), + num_localities(localities_.size()), + width(in_width), + height(num_localities / width) + { + if (num_localities % width != 0) + { + throw std::invalid_argument("num_localities must be divisible by width"); + } + } + + std::size_t num_localities; + std::size_t width; + std::size_t height; +}; + +namespace schedule +{ + +constexpr std::size_t +covariance_tile(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +{ + return (row % sched.height) + (col % sched.width); +} + +constexpr std::size_t +cross_covariance_tile(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +{ + return (row % sched.height) + (col % sched.width); +} + +constexpr std::size_t alpha_tile(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t i) +{ + return (i % sched.height) + (i % sched.width); +} + +constexpr std::size_t prediction_tile(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t i) +{ + return (i % sched.height) + (i % sched.width); +} + +constexpr std::size_t cholesky_potrf(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k) +{ + return (k % sched.height) + (k % sched.width); +} + +constexpr std::size_t cholesky_syrk(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t m) +{ + return (m % sched.height) + (m % sched.width); +} + +constexpr std::size_t +cholesky_trsm(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) +{ + return (m % sched.height) + (k % sched.width); +} + +constexpr std::size_t +cholesky_gemm(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m, std::size_t n) +{ + return (m % sched.height) + (n % sched.width); +} + +constexpr std::size_t solve_trsv(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +{ + return (k % sched.height) + (k % sched.width); +} + +constexpr std::size_t solve_trsm(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k) +{ + return (k % sched.height) + (k % sched.width); +} + +constexpr std::size_t +solve_gemv(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) +{ + return (k % sched.height) + (m % sched.width); +} + +constexpr std::size_t +solve_matrix_trsm(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t c, std::size_t k) +{ + return (k % sched.height) + (c % sched.width); +} + +constexpr std::size_t solve_matrix_gemm( + const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t c, std::size_t k, std::size_t m) +{ + return (m % sched.height) + (c % sched.width); +} + +constexpr std::size_t +multiply_gemv(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) +{ + return (k % sched.height) + (m % sched.width); +} + +constexpr std::size_t k_rank_dot_diag_syrk(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k) +{ + return (k % sched.height) + (k % sched.width); +} + +constexpr std::size_t +k_rank_gemm(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) +{ + return (k * n_tiles + m) % sched.num_localities; +} + +constexpr std::size_t vector_axpy(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +{ + return (k * n_tiles + k) % sched.num_localities; +} + +constexpr std::size_t get_diagonal(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +{ + return (k * n_tiles + k) % sched.num_localities; +} + +constexpr std::size_t compute_loss(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +{ + return (k * n_tiles + k) % sched.num_localities; +} + +} // namespace schedule + +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/scheduler/sma.hpp b/core/include/gprat/scheduler/sma.hpp new file mode 100644 index 00000000..e43e9d1d --- /dev/null +++ b/core/include/gprat/scheduler/sma.hpp @@ -0,0 +1,175 @@ +#ifndef GPRAT_SCHEDULER_SMA_HPP +#define GPRAT_SCHEDULER_SMA_HPP + +#pragma once + +#include "gprat/detail/actions.hpp" +#include "gprat/detail/config.hpp" +#include "gprat/scheduler.hpp" + +GPRAT_NS_BEGIN + +struct tiled_scheduler_sma : tiled_scheduler_distributed +{ + using tiled_scheduler_distributed::tiled_scheduler_distributed; + + std::size_t num_localities = localities_.size(); +}; + +namespace schedule +{ + +constexpr std::size_t +covariance_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +{ + return (row + col) % sched.num_localities; +} + +constexpr std::size_t +cross_covariance_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +{ + return (row + col) % sched.num_localities; +} + +constexpr std::size_t alpha_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) +{ + return (2 * i) % sched.num_localities; +} + +constexpr std::size_t prediction_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) +{ + return (2 * i) % sched.num_localities; +} + +constexpr std::size_t +t_cross_covariance_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +{ + return (row + col) % sched.num_localities; +} + +constexpr std::size_t +prior_K_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +{ + return (row + col) % sched.num_localities; +} + +constexpr std::size_t +K_inv_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +{ + return (row + col) % sched.num_localities; +} + +constexpr std::size_t +K_grad_v_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +{ + return (row + col) % sched.num_localities; +} + +constexpr std::size_t +K_grad_l_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +{ + return (row + col) % sched.num_localities; +} + +constexpr std::size_t uncertainty_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) +{ + return (2 * i) % sched.num_localities; +} + +constexpr std::size_t inter_alpha_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) +{ + return (2 * i) % sched.num_localities; +} + +constexpr std::size_t diag_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) +{ + return i % sched.num_localities; +} + +constexpr std::size_t cholesky_potrf(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) +{ + return (2 * k) % sched.num_localities; +} + +constexpr std::size_t cholesky_syrk(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t m) +{ + return (2 * m) % sched.num_localities; +} + +constexpr std::size_t +cholesky_trsm(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) +{ + return (k + m) % sched.num_localities; +} + +constexpr std::size_t +cholesky_gemm(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m, std::size_t n) +{ + return (m + n) % sched.num_localities; +} + +constexpr std::size_t solve_trsv(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) +{ + return (2 * k) % sched.num_localities; +} + +constexpr std::size_t solve_trsm(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) +{ + return (2 * k) % sched.num_localities; +} + +constexpr std::size_t +solve_gemv(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) +{ + return (k + m) % sched.num_localities; +} + +constexpr std::size_t +solve_matrix_trsm(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t c, std::size_t k) +{ + return (k + c) % sched.num_localities; +} + +constexpr std::size_t solve_matrix_gemm( + const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t c, std::size_t /*k*/, std::size_t m) +{ + return (c + m) % sched.num_localities; +} + +constexpr std::size_t +multiply_gemv(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) +{ + return (k + m) % sched.num_localities; +} + +constexpr std::size_t k_rank_dot_diag_syrk(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) +{ + return (2 * k) % sched.num_localities; +} + +constexpr std::size_t +k_rank_gemm(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t /*c*/, std::size_t k, std::size_t m) +{ + return (k + m) % sched.num_localities; +} + +constexpr std::size_t vector_axpy(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) +{ + return (2 * k) % sched.num_localities; +} + +constexpr std::size_t get_diagonal(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) +{ + return (2 * k) % sched.num_localities; +} + +constexpr std::size_t compute_loss(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) +{ + return (2 * k) % sched.num_localities; +} + +} // namespace schedule + +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/tile_cache.hpp b/core/include/gprat/tile_cache.hpp new file mode 100644 index 00000000..6410b0b5 --- /dev/null +++ b/core/include/gprat/tile_cache.hpp @@ -0,0 +1,97 @@ +#ifndef GPRAT_TILECACHE_HPP +#define GPRAT_TILECACHE_HPP + +#pragma once + +#include "gprat/tile_data.hpp" + +#include +#include + +GPRAT_NS_BEGIN + +namespace detail +{ +hpx::util::cache::statistics::local_full_statistics &get_global_statistics(); + +/// @brief Statistics implementation that uses counters shared between all tile_cache instances +class global_full_statistics +{ + public: + using update_on_exit = hpx::util::cache::statistics::local_full_statistics::update_on_exit; + + // ReSharper disable once CppNonExplicitConversionOperator + operator hpx::util::cache::statistics::local_full_statistics &() const { return get_global_statistics(); } + + void got_hit() noexcept { get_global_statistics().got_hit(); } + + void got_miss() noexcept { get_global_statistics().got_miss(); } + + void got_insertion() noexcept { get_global_statistics().got_insertion(); } + + void got_eviction() noexcept { get_global_statistics().got_eviction(); } + + void clear() noexcept { get_global_statistics().clear(); } +}; +} // namespace detail + +/** + * @brief LRU cache for mutable_tile_data objects with versioning support + * @tparam T Tile data type. + */ +template +class tile_cache +{ + friend struct tile_cache_counters; + + public: + explicit tile_cache(std::size_t max_size = 16) : + cache_(max_size) + { } + + bool try_get(const hpx::naming::gid_type &key, std::size_t generation, mutable_tile_data &cached_data) + { + std::lock_guard g(mutex_); + + entry e; + { + hpx::naming::gid_type unused; + if (!cache_.get_entry(key, unused, e)) + { + return false; + } + } + + if (e.generation == generation) + { + cached_data = e.data; + return true; + } + + // Erase the obsolete entry + cache_.erase([&](const auto &p) { return p.first == key; }); + return false; + } + + void insert(const hpx::naming::gid_type &key, std::size_t generation, const mutable_tile_data &data) + { + std::lock_guard g(mutex_); + cache_.insert(key, entry{ data, generation }); + } + + void clear() { cache_.clear(); } + + private: + struct entry + { + mutable_tile_data data; + std::size_t generation = 0; + }; + + hpx::mutex mutex_; // lru_cache is not thread-safe! + hpx::util::cache::lru_cache cache_; +}; + +GPRAT_NS_END + +#endif diff --git a/core/include/gprat/tile_data.hpp b/core/include/gprat/tile_data.hpp index 96fd3f3c..c4b2c086 100644 --- a/core/include/gprat/tile_data.hpp +++ b/core/include/gprat/tile_data.hpp @@ -105,7 +105,9 @@ class const_tile_data [[nodiscard]] const T *data() const { if (!cpu_data_.data()) + { throw std::runtime_error("no data"); + } return cpu_data_.data(); } @@ -155,9 +157,14 @@ class mutable_tile_data : public const_tile_data public: using const_tile_data::const_tile_data; - [[nodiscard]] T *data() const { + [[nodiscard]] T *data() const + { if (!this->cpu_data_.data()) - throw std::runtime_error("no data");return const_cast(this->cpu_data_.data()); } + { + throw std::runtime_error("no data"); + } + return const_cast(this->cpu_data_.data()); + } [[nodiscard]] T *begin() const noexcept { return const_cast(this->cpu_data_.data()); } diff --git a/examples/distributed/src/distributed_tile.hpp b/core/include/gprat/tiled_dataset.hpp similarity index 81% rename from examples/distributed/src/distributed_tile.hpp rename to core/include/gprat/tiled_dataset.hpp index 935341f0..052b204e 100644 --- a/examples/distributed/src/distributed_tile.hpp +++ b/core/include/gprat/tiled_dataset.hpp @@ -1,106 +1,26 @@ -#pragma once +#ifndef GPRAT_COMPONENTS_TILED_DATASET_HPP +#define GPRAT_COMPONENTS_TILED_DATASET_HPP +#pragma once + +#include "gprat/detail/actions.hpp" +#include "gprat/detail/config.hpp" +#include "gprat/performance_counters.hpp" +#include "gprat/tile_cache.hpp" #include "gprat/tile_data.hpp" -#include #include #include -#include #include #include #include #include #include -#include #include #include GPRAT_NS_BEGIN -void register_distributed_tile_counters(); -void record_transmission_time(std::int64_t elapsed_ns); - -void track_tile_server_allocation(std::size_t size); -void track_tile_server_deallocation(std::size_t size); - -namespace detail -{ -hpx::util::cache::statistics::local_full_statistics &get_global_statistics(); - -/////////////////////////////////////////////////////////////////////////// -class global_full_statistics -{ - public: - using update_on_exit = hpx::util::cache::statistics::local_full_statistics::update_on_exit; - - // ReSharper disable once CppNonExplicitConversionOperator - operator hpx::util::cache::statistics::local_full_statistics &() const { return get_global_statistics(); } - - void got_hit() noexcept { get_global_statistics().got_hit(); } - - void got_miss() noexcept { get_global_statistics().got_miss(); } - - void got_insertion() noexcept { get_global_statistics().got_insertion(); } - - void got_eviction() noexcept { get_global_statistics().got_eviction(); } - - void clear() noexcept { get_global_statistics().clear(); } -}; -} // namespace detail - -template -class tile_cache -{ - friend struct tile_cache_counters; - - public: - tile_cache() : - cache_(16) - { } - - bool try_get(const hpx::naming::gid_type &key, std::size_t generation, mutable_tile_data &cached_data) - { - std::lock_guard g(mutex_); - - entry e; - { - hpx::naming::gid_type unused; - if (!cache_.get_entry(key, unused, e)) - { - return false; - } - } - - if (e.generation == generation) - { - cached_data = e.data; - return true; - } - - // Erase the obsolete entry - cache_.erase([&](const auto &p) { return p.first == key; }); - return false; - } - - void insert(const hpx::naming::gid_type &key, std::size_t generation, const mutable_tile_data &data) - { - std::lock_guard g(mutex_); - cache_.insert(key, entry{ data, generation }); - } - - void clear() { cache_.clear(); } - - private: - struct entry - { - mutable_tile_data data; - std::size_t generation = 0; - }; - - hpx::mutex mutex_; - hpx::util::cache::lru_cache cache_; -}; - namespace server { @@ -259,7 +179,7 @@ struct tile_manager : hpx::components::component_base> record_transmission_time(timer.elapsed_nanoseconds()); auto data = f.get(); cache_.insert(gid, generation, data); - self = {}; // release our reference + self = {}; // release our reference return data; }); } @@ -319,9 +239,6 @@ struct tile_manager : hpx::components::component_base> typedef ::GPRAT_NS::server::tile_manager HPX_PP_CAT(_server_tile_manager_, HPX_PP_CAT(type, name)); \ GPRAT_REGISTER_TILE_MANAGER_IMPL(HPX_PP_CAT(_server_tile_manager_, HPX_PP_CAT(type, name)), name) -template -class tiled_dataset_accessor; - template class tile_handle { @@ -481,4 +398,30 @@ create_tiled_dataset(std::span> targe return tiles; } +template +tiled_dataset make_tiled_dataset(const tiled_scheduler_distributed &sched, std::size_t num_tiles, Mapper &&mapper) +{ + const auto num_localities = sched.localities_.size(); + std::vector> targets; + targets.reserve(num_localities); + + for (std::size_t i = 0; i < num_localities; ++i) + { + targets.emplace_back(sched.localities_[i], 0); + } + + for (std::size_t i = 0; i < num_tiles; i++) + { + ++targets[mapper(i) % num_localities].second; + } + + return create_tiled_dataset(targets, num_tiles); +} + GPRAT_NS_END + +// Register the double version by default +// Users can register custom types in the same way +GPRAT_REGISTER_TILED_DATASET_DECLARATION(double, double); + +#endif diff --git a/examples/distributed/src/distributed_blas.cpp b/core/src/cpu/adapter_cblas_fp64_actions.cpp similarity index 80% rename from examples/distributed/src/distributed_blas.cpp rename to core/src/cpu/adapter_cblas_fp64_actions.cpp index 0e0af8b4..64c019ab 100644 --- a/examples/distributed/src/distributed_blas.cpp +++ b/core/src/cpu/adapter_cblas_fp64_actions.cpp @@ -1,31 +1,21 @@ -#include "distributed_blas.hpp" - -#include "gprat/cpu/adapter_cblas_fp64.hpp" +#include "gprat/cpu/adapter_cblas_fp64_actions.hpp" #include -HPX_REGISTER_ACTION(GPRAT_NS::potrf_distributed_action); -HPX_REGISTER_ACTION(GPRAT_NS::trsm_distributed_action); -HPX_REGISTER_ACTION(GPRAT_NS::syrk_distributed_action); -HPX_REGISTER_ACTION(GPRAT_NS::gemm_distributed_action); -HPX_REGISTER_ACTION(GPRAT_NS::trsv_distributed_action); -HPX_REGISTER_ACTION(GPRAT_NS::gemv_distributed_action); -HPX_REGISTER_ACTION(GPRAT_NS::dot_diag_syrk_distributed_action); -HPX_REGISTER_ACTION(GPRAT_NS::dot_diag_gemm_distributed_action); -HPX_REGISTER_ACTION(GPRAT_NS::axpy_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::potrf, GPRAT_NS::cpu::potrf_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::trsm, GPRAT_NS::cpu::trsm_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::syrk, GPRAT_NS::cpu::syrk_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::gemm, GPRAT_NS::cpu::gemm_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::trsv, GPRAT_NS::cpu::trsv_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::gemv, GPRAT_NS::cpu::gemv_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::dot_diag_syrk, GPRAT_NS::cpu::dot_diag_syrk_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::dot_diag_gemm, GPRAT_NS::cpu::dot_diag_gemm_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::axpy, GPRAT_NS::cpu::axpy_distributed_action); GPRAT_NS_BEGIN -GPRAT_DEFINE_PLAIN_ACTION_FOR(&potrf); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&trsm); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&syrk); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&gemm); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&trsv); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&gemv); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&dot_diag_syrk); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&dot_diag_gemm); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&axpy); - +namespace cpu +{ hpx::future> potrf_distributed(const tile_handle &A, int N) { return hpx::dataflow( @@ -147,5 +137,6 @@ hpx::future> axpy_distributed(const tile_handle &y, y.get_async(), x.get_async()); } +} // namespace cpu GPRAT_NS_END diff --git a/core/src/cpu/gp_algorithms_actions.cpp b/core/src/cpu/gp_algorithms_actions.cpp new file mode 100644 index 00000000..8fdb12d8 --- /dev/null +++ b/core/src/cpu/gp_algorithms_actions.cpp @@ -0,0 +1,100 @@ +#include "gprat/cpu/gp_algorithms_actions.hpp" + +#include + +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_covariance, + GPRAT_NS::cpu::gen_tile_covariance_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_prior_covariance, + GPRAT_NS::cpu::gen_tile_prior_covariance_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_prior_covariance, + GPRAT_NS::cpu::gen_tile_full_prior_covariance_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_cross_covariance, + GPRAT_NS::cpu::gen_tile_cross_covariance_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_transpose, GPRAT_NS::cpu::gen_tile_transpose_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_output, GPRAT_NS::cpu::gen_tile_output_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_zeros, GPRAT_NS::cpu::gen_tile_zeros_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_identity, GPRAT_NS::cpu::gen_tile_identity_distributed_action); + +GPRAT_NS_BEGIN + +namespace cpu +{ +hpx::future> gen_tile_covariance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &input) +{ + return tile.set_async(cpu::gen_tile_covariance(row, col, N, n_regressors, sek_params, input)); +} + +hpx::future> gen_tile_prior_covariance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &input) +{ + return tile.set_async(cpu::gen_tile_prior_covariance(row, col, N, n_regressors, sek_params, input)); +} + +hpx::future> gen_tile_full_prior_covariance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &input) +{ + return tile.set_async(cpu::gen_tile_full_prior_covariance(row, col, N, n_regressors, sek_params, input)); +} + +hpx::future> gen_tile_cross_covariance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N_row, + std::size_t N_col, + std::size_t n_regressors, + const SEKParams &sek_params, + const std::vector &row_input, + const std::vector &col_input) +{ + return tile.set_async( + cpu::gen_tile_cross_covariance(row, col, N_row, N_col, n_regressors, sek_params, row_input, col_input)); +} + +hpx::future> gen_tile_transpose_distributed( + const tile_handle &tile, std::size_t N_row, std::size_t N_col, const tile_handle &src) +{ + return hpx::dataflow( + hpx::launch::async, + [=](hpx::future> &&tiled) + { return tile.set_async(cpu::gen_tile_transpose(N_row, N_col, tiled.get())); }, + src.get_async()); +} + +hpx::future> gen_tile_output_distributed( + const tile_handle &tile, std::size_t row, std::size_t N, const std::vector &output) +{ + return tile.set_async(cpu::gen_tile_output(row, N, output)); +} + +hpx::future> gen_tile_zeros_distributed(const tile_handle &tile, std::size_t N) +{ + return tile.set_async(cpu::gen_tile_zeros(N)); +} + +hpx::future> gen_tile_identity_distributed(const tile_handle &tile, std::size_t N) +{ + return tile.set_async(cpu::gen_tile_identity(N)); +} +} // namespace cpu + +GPRAT_NS_END diff --git a/core/src/cpu/gp_optimizer_actions.cpp b/core/src/cpu/gp_optimizer_actions.cpp new file mode 100644 index 00000000..df0b2c4d --- /dev/null +++ b/core/src/cpu/gp_optimizer_actions.cpp @@ -0,0 +1,93 @@ +#include "gprat/cpu/gp_optimizer_actions.hpp" + +#include + +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_covariance_with_distance, + GPRAT_NS::cpu::gen_tile_covariance_with_distance_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_grad_l, GPRAT_NS::cpu::gen_tile_grad_l_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::gen_tile_grad_v, GPRAT_NS::cpu::gen_tile_grad_v_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::compute_loss, GPRAT_NS::cpu::compute_loss_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::compute_trace, GPRAT_NS::cpu::compute_trace_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::compute_dot, GPRAT_NS::cpu::compute_dot_distributed_action); +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::compute_trace_diag, GPRAT_NS::cpu::compute_trace_diag_distributed_action); + +GPRAT_NS_BEGIN + +namespace cpu +{ + +hpx::future> gen_tile_covariance_with_distance_distributed( + const tile_handle &tile, + std::size_t row, + std::size_t col, + std::size_t N, + const SEKParams &sek_params, + const const_tile_data &distance) +{ + return tile.set_async(cpu::gen_tile_covariance_with_distance(row, col, N, sek_params, distance)); +} + +hpx::future> gen_tile_grad_l_distributed( + const tile_handle &tile, + std::size_t N, + const SEKParams &sek_params, + const const_tile_data &distance) +{ + return tile.set_async(cpu::gen_tile_grad_l(N, sek_params, distance)); +} + +hpx::future> gen_tile_grad_v_distributed( + const tile_handle &tile, + std::size_t N, + const SEKParams &sek_params, + const const_tile_data &distance) +{ + return tile.set_async(cpu::gen_tile_grad_v(N, sek_params, distance)); +} + +hpx::future compute_loss_distributed(const tile_handle &K_diag_tile, + const tile_handle &alpha_tile, + const tile_handle &y_tile, + std::size_t N) +{ + return hpx::dataflow( + hpx::launch::async, + [=](hpx::future> &&K_diag_tiled, + hpx::future> &&alpha_tiled, + hpx::future> &&y_tiled) + { return cpu::compute_loss(K_diag_tiled.get(), alpha_tiled.get(), y_tiled.get(), N); }, + K_diag_tile.get_async(), + alpha_tile.get_async(), + y_tile.get_async()); +} + +hpx::future compute_trace_distributed(const tile_handle &diagonal, double trace) +{ + return hpx::dataflow( + hpx::launch::async, + [=](hpx::future> &&diagonald) { return cpu::compute_trace(diagonald.get(), trace); }, + diagonal.get_async()); +} + +hpx::future +compute_dot_distributed(const tile_handle &vector_T, const tile_handle &vector, double result) +{ + return hpx::dataflow( + hpx::launch::async, + [=](hpx::future> &&vector_Td, hpx::future> &&vectord) + { return cpu::compute_dot(vector_Td.get(), vectord.get(), result); }, + vector_T.get_async(), + vector.get_async()); +} + +hpx::future compute_trace_diag_distributed(const tile_handle &tile, double trace, std::size_t N) +{ + return hpx::dataflow( + hpx::launch::async, + [=](hpx::future> &&tiled) { return cpu::compute_trace_diag(tiled.get(), trace, N); }, + tile.get_async()); +} + +} // namespace cpu + +GPRAT_NS_END diff --git a/core/src/cpu/gp_uncertainty_actions.cpp b/core/src/cpu/gp_uncertainty_actions.cpp new file mode 100644 index 00000000..466fd396 --- /dev/null +++ b/core/src/cpu/gp_uncertainty_actions.cpp @@ -0,0 +1,23 @@ +#include "gprat/cpu/gp_uncertainty_actions.hpp" + +#include + +GPRAT_DEFINE_PLAIN_ACTION_FOR(&GPRAT_NS::cpu::get_matrix_diagonal, + GPRAT_NS::cpu::get_matrix_diagonal_distributed_action); + +GPRAT_NS_BEGIN + +namespace cpu +{ +hpx::future> get_matrix_diagonal_distributed(const tile_handle &A, std::size_t M) +{ + return hpx::dataflow( + hpx::launch::async, + [A, M](hpx::future> &&Ad) + { return A.set_async(cpu::get_matrix_diagonal(Ad.get(), M)); }, + A.get_async()); +} + +} // namespace cpu + +GPRAT_NS_END diff --git a/core/src/gprat.cpp b/core/src/gprat.cpp index 969fdb9e..54480176 100644 --- a/core/src/gprat.cpp +++ b/core/src/gprat.cpp @@ -1,6 +1,7 @@ #include "gprat/gprat.hpp" #include "gprat/cpu/gp_functions.hpp" +#include "gprat/tiled_dataset.hpp" #include "gprat/utils.hpp" #if GPRAT_WITH_CUDA diff --git a/core/src/performance_counters.cpp b/core/src/performance_counters.cpp index 0434e2bb..42e51989 100644 --- a/core/src/performance_counters.cpp +++ b/core/src/performance_counters.cpp @@ -1,5 +1,7 @@ #include "gprat/performance_counters.hpp" +#include "gprat/tile_cache.hpp" + #include #include #include @@ -15,6 +17,10 @@ GPRAT_NS_BEGIN GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_data_allocations) GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_data_deallocations) +GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_server_allocations) +GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_server_deallocations) +GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_transmission_time) +GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_transmission_count) #undef GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR @@ -22,6 +28,20 @@ void track_tile_data_allocation(std::size_t /*size*/) { tile_data_allocations += void track_tile_data_deallocation(std::size_t /*size*/) { tile_data_deallocations += 1; } +void track_tile_server_allocation(std::size_t /*size*/) { tile_server_allocations += 1; } + +void track_tile_server_deallocation(std::size_t /*size*/) { tile_server_deallocations += 1; } + +void record_transmission_time(std::int64_t elapsed_ns) +{ + HPX_ASSERT(elapsed_ns >= 0); + tile_transmission_count += 1; + if (elapsed_ns > 0) + { + tile_transmission_time += static_cast(elapsed_ns); + } +} + #ifdef HPX_HAVE_MODULE_PERFORMANCE_COUNTERS // These are non-public functions of their respective CUs. namespace detail @@ -43,6 +63,26 @@ void register_performance_counters() GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/tile_data/num_allocations", tile_data_allocations); GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/tile_data/num_deallocations", tile_data_deallocations); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/tile_server/num_allocations", tile_server_allocations); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/tile_server/num_deallocations", tile_server_deallocations); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/tile_cache/transmission_time", tile_transmission_time); + GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR("/gprat/tile_cache/transmission_count", tile_transmission_count); + +#undef GPRAT_MAKE_STATISTICS_ACCESSOR + + // XXX: you can do this with templates, but it's quite a bit more complicated +#define GPRAT_MAKE_STATISTICS_ACCESSOR(name, stats_expr) \ + hpx::performance_counters::install_counter_type( \ + name, \ + [](bool reset) { return (stats_expr) (reset); }, \ + #stats_expr, \ + "", \ + hpx::performance_counters::counter_type::monotonically_increasing) + + GPRAT_MAKE_STATISTICS_ACCESSOR("/gprat/tile_cache/hits", detail::get_global_statistics().hits); + GPRAT_MAKE_STATISTICS_ACCESSOR("/gprat/tile_cache/misses", detail::get_global_statistics().misses); + GPRAT_MAKE_STATISTICS_ACCESSOR("/gprat/tile_cache/evictions", detail::get_global_statistics().evictions); + GPRAT_MAKE_STATISTICS_ACCESSOR("/gprat/tile_cache/insertions", detail::get_global_statistics().insertions); #undef GPRAT_MAKE_STATISTICS_ACCESSOR diff --git a/core/src/tile_cache.cpp b/core/src/tile_cache.cpp new file mode 100644 index 00000000..6a1c658a --- /dev/null +++ b/core/src/tile_cache.cpp @@ -0,0 +1,15 @@ +#include "gprat/tile_cache.hpp" + +GPRAT_NS_BEGIN + +namespace detail +{ +hpx::util::cache::statistics::local_full_statistics &get_global_statistics() +{ + static hpx::util::cache::statistics::local_full_statistics stats; + return stats; +} + +} // namespace detail + +GPRAT_NS_END diff --git a/core/src/tiled_dataset.cpp b/core/src/tiled_dataset.cpp new file mode 100644 index 00000000..4f8df182 --- /dev/null +++ b/core/src/tiled_dataset.cpp @@ -0,0 +1,3 @@ +#include "gprat/tiled_dataset.hpp" + +GPRAT_REGISTER_TILED_DATASET(double, double); diff --git a/examples/distributed/CMakeLists.txt b/examples/distributed/CMakeLists.txt index 8abff6e8..c2b56c76 100644 --- a/examples/distributed/CMakeLists.txt +++ b/examples/distributed/CMakeLists.txt @@ -1,19 +1,18 @@ -add_executable(gprat_distributed src/main.cpp src/distributed_blas.cpp - src/distributed_tile.cpp) +add_executable(gprat_distributed src/main.cpp) target_compile_features(gprat_distributed PUBLIC cxx_std_20) include(FetchContent) FetchContent_Declare( - Catch2 - GIT_REPOSITORY https://github.com/catchorg/Catch2.git - GIT_TAG v3.8.0) + Catch2 + GIT_REPOSITORY https://github.com/catchorg/Catch2.git + GIT_TAG v3.8.0) FetchContent_MakeAvailable(Catch2) find_package(Boost REQUIRED) -target_link_libraries(gprat_distributed PUBLIC GPRat::core HPX::hpx Catch2::Catch2 - Boost::boost) +target_link_libraries(gprat_distributed PUBLIC GPRat::core HPX::hpx + Catch2::Catch2 Boost::boost) set_target_properties(gprat_distributed PROPERTIES VS_DEBUGGER_WORKING_DIRECTORY "${CMAKE_SOURCE_DIR}") diff --git a/examples/distributed/src/distributed_cholesky.hpp b/examples/distributed/src/distributed_cholesky.hpp deleted file mode 100644 index 79207413..00000000 --- a/examples/distributed/src/distributed_cholesky.hpp +++ /dev/null @@ -1,316 +0,0 @@ -#pragma once - -#include "gprat/scheduler.hpp" - -#include "distributed_tile.hpp" -#include "scheduling.hpp" - -GPRAT_NS_BEGIN - -struct tiled_scheduler_sma : tiled_scheduler_distributed -{ - using tiled_scheduler_distributed::tiled_scheduler_distributed; - - std::size_t num_localities = localities_.size(); -}; - -struct tiled_scheduler_cyclic : tiled_scheduler_distributed -{ - using tiled_scheduler_distributed::tiled_scheduler_distributed; - - /// @brief Create a new scheduler that targets all localities. - explicit tiled_scheduler_cyclic(std::size_t in_width = 1) : - num_localities(localities_.size()), - width(in_width), - height(num_localities / width) - { - if (num_localities % width != 0) - { - throw std::invalid_argument("num_localities must be divisible by width"); - } - } - - /// @brief Create a new scheduler that targets the given localities. - explicit tiled_scheduler_cyclic(std::vector in_localities, std::size_t in_width = 1) : - tiled_scheduler_distributed(std::move(in_localities)), - num_localities(localities_.size()), - width(in_width), - height(num_localities / width) - { - if (num_localities % width != 0) - { - throw std::invalid_argument("num_localities must be divisible by width"); - } - } - - std::size_t num_localities; - std::size_t width; - std::size_t height; -}; - -namespace schedule -{ - -#ifdef _MSC_VER -#pragma warning(push) -#pragma warning(disable : 4100) -#endif - -constexpr std::size_t -covariance_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t row, std::size_t col) -{ - return (row + col) % sched.num_localities; -} - -constexpr std::size_t -cross_covariance_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t row, std::size_t col) -{ - return (row + col) % sched.num_localities; -} - -constexpr std::size_t alpha_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t i) -{ - return (2 * i) % sched.num_localities; -} - -constexpr std::size_t prediction_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t i) -{ - return (2 * i) % sched.num_localities; -} - -constexpr std::size_t -t_cross_covariance_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t row, std::size_t col) -{ - return (row + col) % sched.num_localities; -} - -constexpr std::size_t -prior_K_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t row, std::size_t col) -{ - return (row + col) % sched.num_localities; -} - -constexpr std::size_t -K_inv_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t row, std::size_t col) -{ - return (row + col) % sched.num_localities; -} - -constexpr std::size_t -K_grad_v_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t row, std::size_t col) -{ - return (row + col) % sched.num_localities; -} - -constexpr std::size_t -K_grad_l_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t row, std::size_t col) -{ - return (row + col) % sched.num_localities; -} - -constexpr std::size_t uncertainty_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t i) -{ - return (2 * i) % sched.num_localities; -} - -constexpr std::size_t inter_alpha_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t i) -{ - return (2 * i) % sched.num_localities; -} - -constexpr std::size_t diag_tile(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t i) -{ - return i % sched.num_localities; -} - -constexpr std::size_t cholesky_potrf(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k) -{ - return (2 * k) % sched.num_localities; -} - -constexpr std::size_t cholesky_syrk(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t m) -{ - return (2 * m) % sched.num_localities; -} - -constexpr std::size_t cholesky_trsm(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k, std::size_t m) -{ - return (k + m) % sched.num_localities; -} - -constexpr std::size_t -cholesky_gemm(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k, std::size_t m, std::size_t n) -{ - return (m + n) % sched.num_localities; -} - -constexpr std::size_t solve_trsv(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k) -{ - return (2 * k) % sched.num_localities; -} - -constexpr std::size_t solve_trsm(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k) -{ - return (2 * k) % sched.num_localities; -} - -constexpr std::size_t solve_gemv(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k, std::size_t m) -{ - return (k + m) % sched.num_localities; -} - -constexpr std::size_t -solve_matrix_trsm(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t c, std::size_t k) -{ - return (k + c) % sched.num_localities; -} - -constexpr std::size_t -solve_matrix_gemm(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) -{ - return (c + m) % sched.num_localities; -} - -constexpr std::size_t multiply_gemv(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k, std::size_t m) -{ - return (k + m) % sched.num_localities; -} - -constexpr std::size_t k_rank_dot_diag_syrk(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k) -{ - return (2 * k) % sched.num_localities; -} - -constexpr std::size_t -k_rank_gemm(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) -{ - return (k + m) % sched.num_localities; -} - -constexpr std::size_t vector_axpy(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k) -{ - return (2 * k) % sched.num_localities; -} - -constexpr std::size_t get_diagonal(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k) -{ - return (2 * k) % sched.num_localities; -} - -constexpr std::size_t compute_loss(const tiled_scheduler_sma &sched, std::size_t n_tiles, std::size_t k) -{ - return (2 * k) % sched.num_localities; -} - -// ========================== - -constexpr std::size_t -covariance_tile(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t row, std::size_t col) -{ - return (row % sched.height) + (col % sched.width); -} - -constexpr std::size_t -cross_covariance_tile(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t row, std::size_t col) -{ - return (row % sched.height) + (col % sched.width); -} - -constexpr std::size_t alpha_tile(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t i) -{ - return (i % sched.height) + (i % sched.width); -} - -constexpr std::size_t prediction_tile(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t i) -{ - return (i % sched.height) + (i % sched.width); -} - -constexpr std::size_t cholesky_potrf(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) -{ - return (k % sched.height) + (k % sched.width); -} - -constexpr std::size_t cholesky_syrk(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t m) -{ - return (m % sched.height) + (m % sched.width); -} - -constexpr std::size_t -cholesky_trsm(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k, std::size_t m) -{ - return (m % sched.height) + (k % sched.width); -} - -constexpr std::size_t -cholesky_gemm(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k, std::size_t m, std::size_t n) -{ - return (m % sched.height) + (n % sched.width); -} - -constexpr std::size_t solve_trsv(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) -{ - return (k % sched.height) + (k % sched.width); -} - -constexpr std::size_t solve_trsm(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) -{ - return (k % sched.height) + (k % sched.width); -} - -constexpr std::size_t solve_gemv(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k, std::size_t m) -{ - return (k % sched.height) + (m % sched.width); -} - -constexpr std::size_t -solve_matrix_trsm(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t c, std::size_t k) -{ - return (k % sched.height) + (c % sched.width); -} - -constexpr std::size_t -solve_matrix_gemm(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) -{ - return (m % sched.height) + (c % sched.width); -} - -constexpr std::size_t -multiply_gemv(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k, std::size_t m) -{ - return (k % sched.height) + (m % sched.width); -} - -constexpr std::size_t k_rank_dot_diag_syrk(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) -{ - return (k % sched.height) + (k % sched.width); -} - -constexpr std::size_t -k_rank_gemm(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) -{ - return (k * n_tiles + m) % sched.num_localities; -} - -constexpr std::size_t vector_axpy(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) -{ - return (k * n_tiles + k) % sched.num_localities; -} - -constexpr std::size_t get_diagonal(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) -{ - return (k * n_tiles + k) % sched.num_localities; -} - -constexpr std::size_t compute_loss(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) -{ - return (k * n_tiles + k) % sched.num_localities; -} - -#ifdef _MSC_VER -#pragma warning(pop) -#endif - -} // namespace schedule - -GPRAT_NS_END diff --git a/examples/distributed/src/distributed_tile.cpp b/examples/distributed/src/distributed_tile.cpp deleted file mode 100644 index 5b4e1910..00000000 --- a/examples/distributed/src/distributed_tile.cpp +++ /dev/null @@ -1,95 +0,0 @@ -#include "distributed_tile.hpp" - -#include -#include - -GPRAT_NS_BEGIN - -namespace detail -{ -hpx::util::cache::statistics::local_full_statistics &get_global_statistics() -{ - static hpx::util::cache::statistics::local_full_statistics stats; - return stats; -} - -} // namespace detail - -std::atomic tile_transmission_time(0); -std::atomic tile_transmission_count(0); -std::atomic tile_data_allocations(0); -std::atomic tile_data_deallocations(0); -std::atomic tile_server_allocations(0); -std::atomic tile_server_deallocations(0); - -void record_transmission_time(std::int64_t elapsed_ns) -{ - HPX_ASSERT(elapsed_ns >= 0); - tile_transmission_count += 1; - if (elapsed_ns > 0) - { - tile_transmission_time += static_cast(elapsed_ns); - } -} - -void track_tile_server_allocation(std::size_t /*size*/) { tile_server_allocations += 1; } - -void track_tile_server_deallocation(std::size_t /*size*/) { tile_server_deallocations += 1; } - -#define GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(name) \ - std::uint64_t get_##name(bool reset) { return hpx::util::get_and_reset_value(name, reset); } - -GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_transmission_count) -GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_transmission_time) -GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_server_allocations) -GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR(tile_server_deallocations) - -#undef GPRAT_MAKE_SIMPLE_COUNTER_ACCESSOR - -void register_distributed_tile_counters() -{ - // XXX: you can do this with templates, but it's quite a bit more complicated -#define GPRAT_MAKE_STATISTICS_ACCESSOR(name, stats_expr) \ - hpx::performance_counters::install_counter_type( \ - name, \ - [](bool reset) { return (stats_expr) (reset); }, \ - #stats_expr, \ - "", \ - hpx::performance_counters::counter_type::monotonically_increasing) - - GPRAT_MAKE_STATISTICS_ACCESSOR("/gprat/tile_cache/hits", detail::get_global_statistics().hits); - GPRAT_MAKE_STATISTICS_ACCESSOR("/gprat/tile_cache/misses", detail::get_global_statistics().misses); - GPRAT_MAKE_STATISTICS_ACCESSOR("/gprat/tile_cache/evictions", detail::get_global_statistics().evictions); - GPRAT_MAKE_STATISTICS_ACCESSOR("/gprat/tile_cache/insertions", detail::get_global_statistics().insertions); - -#undef GPRAT_MAKE_STATISTICS_ACCESSOR - - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/transmission_count", - &get_tile_transmission_count, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - - hpx::performance_counters::install_counter_type( - "/gprat/tile_cache/transmission_time", - &get_tile_transmission_time, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - - hpx::performance_counters::install_counter_type( - "/gprat/tile_server/num_allocations", - &get_tile_server_allocations, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); - hpx::performance_counters::install_counter_type( - "/gprat/tile_server/num_deallocations", - &get_tile_server_deallocations, - "", - "", - hpx::performance_counters::counter_type::monotonically_increasing); -} - -GPRAT_NS_END diff --git a/examples/distributed/src/main.cpp b/examples/distributed/src/main.cpp index a66d3736..a7b886ec 100644 --- a/examples/distributed/src/main.cpp +++ b/examples/distributed/src/main.cpp @@ -1,13 +1,14 @@ -#include "distributed_blas.hpp" -#include "distributed_cholesky.hpp" -#include "distributed_tile.hpp" - -#include "gprat/cpu/gp_algorithms.hpp" +// All of these are necessary: +#include "gprat/cpu/adapter_cblas_fp64_actions.hpp" +#include "gprat/cpu/gp_algorithms_actions.hpp" #include "gprat/cpu/gp_functions.hpp" +#include "gprat/cpu/gp_optimizer_actions.hpp" +#include "gprat/cpu/gp_uncertainty_actions.hpp" #include "gprat/gprat.hpp" #include "gprat/kernels.hpp" #include "gprat/performance_counters.hpp" -#include "gprat/scheduler.hpp" +#include "gprat/scheduler/sma.hpp" +#include "gprat/tiled_dataset.hpp" #include "gprat/utils.hpp" #include "../../test/src/test_data.hpp" @@ -20,344 +21,12 @@ #include #include -// This is a standalone test, so including this directly is fine. +// This is a standalone example, so including this directly is fine. // Better than having the whole project depend on compiled Boost.Json! #include -GPRAT_REGISTER_TILED_DATASET(double, double); - GPRAT_NS_BEGIN -template -tiled_dataset make_tiled_dataset(const tiled_scheduler_distributed &sched, std::size_t num_tiles, Mapper &&mapper) -{ - const auto num_localities = sched.localities_.size(); - std::vector> targets; - targets.reserve(num_localities); - - for (std::size_t i = 0; i < num_localities; ++i) - { - targets.emplace_back(sched.localities_[i], 0); - } - - for (std::size_t i = 0; i < num_tiles; i++) - { - ++targets[mapper(i) % num_localities].second; - } - - return create_tiled_dataset(targets, num_tiles); -} - -hpx::future> gen_tile_covariance_distributed( - const tile_handle &tile, - std::size_t row, - std::size_t col, - std::size_t N, - std::size_t n_regressors, - const SEKParams &sek_params, - const std::vector &input); -HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_covariance_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_covariance, - gen_tile_covariance_distributed_action, - "gen_tile_covariance"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_covariance); - -hpx::future> gen_tile_covariance_distributed( - const tile_handle &tile, - std::size_t row, - std::size_t col, - std::size_t N, - std::size_t n_regressors, - const SEKParams &sek_params, - const std::vector &input) -{ - return tile.set_async(cpu::gen_tile_covariance(row, col, N, n_regressors, sek_params, input)); -} - -hpx::future> gen_tile_covariance_with_distance_distributed( - const tile_handle &tile, - std::size_t row, - std::size_t col, - std::size_t N, - const SEKParams &sek_params, - const const_tile_data &distance); - -HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_covariance_with_distance_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_covariance_with_distance, - gen_tile_covariance_with_distance_distributed_action, - "gen_tile_covariance_with_distance"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_covariance_with_distance); - -hpx::future> gen_tile_covariance_with_distance_distributed( - const tile_handle &tile, - std::size_t row, - std::size_t col, - std::size_t N, - const SEKParams &sek_params, - const const_tile_data &distance) -{ - return tile.set_async(cpu::gen_tile_covariance_with_distance(row, col, N, sek_params, distance)); -} - -hpx::future> gen_tile_prior_covariance_distributed( - const tile_handle &tile, - std::size_t row, - std::size_t col, - std::size_t N, - std::size_t n_regressors, - const SEKParams &sek_params, - const std::vector &input); -HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_prior_covariance_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_prior_covariance, - gen_tile_prior_covariance_distributed_action, - "gen_tile_prior_covariance"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_prior_covariance); - -hpx::future> gen_tile_prior_covariance_distributed( - const tile_handle &tile, - std::size_t row, - std::size_t col, - std::size_t N, - std::size_t n_regressors, - const SEKParams &sek_params, - const std::vector &input) -{ - return tile.set_async(cpu::gen_tile_prior_covariance(row, col, N, n_regressors, sek_params, input)); -} - -hpx::future> gen_tile_full_prior_covariance_distributed( - const tile_handle &tile, - std::size_t row, - std::size_t col, - std::size_t N, - std::size_t n_regressors, - const SEKParams &sek_params, - const std::vector &input); -HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_full_prior_covariance_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_full_prior_covariance, - gen_tile_full_prior_covariance_distributed_action, - "gen_tile_full_prior_covariance"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_prior_covariance); - -hpx::future> gen_tile_full_prior_covariance_distributed( - const tile_handle &tile, - std::size_t row, - std::size_t col, - std::size_t N, - std::size_t n_regressors, - const SEKParams &sek_params, - const std::vector &input) -{ - return tile.set_async(cpu::gen_tile_full_prior_covariance(row, col, N, n_regressors, sek_params, input)); -} - -hpx::future> gen_tile_cross_covariance_distributed( - const tile_handle &tile, - std::size_t row, - std::size_t col, - std::size_t N_row, - std::size_t N_col, - std::size_t n_regressors, - const SEKParams &sek_params, - const std::vector &row_input, - const std::vector &col_input); - -HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_cross_covariance_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_cross_covariance, - gen_tile_cross_covariance_distributed_action, - "gen_tile_cross_covariance"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_cross_covariance); - -hpx::future> gen_tile_cross_covariance_distributed( - const tile_handle &tile, - std::size_t row, - std::size_t col, - std::size_t N_row, - std::size_t N_col, - std::size_t n_regressors, - const SEKParams &sek_params, - const std::vector &row_input, - const std::vector &col_input) -{ - return tile.set_async( - cpu::gen_tile_cross_covariance(row, col, N_row, N_col, n_regressors, sek_params, row_input, col_input)); -} - -hpx::future> gen_tile_transpose_distributed( - const tile_handle &tile, std::size_t N_row, std::size_t N_col, const tile_handle &src); -HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_transpose_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_transpose, gen_tile_transpose_distributed_action, "gen_tile_transpose"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_transpose); - -hpx::future> gen_tile_transpose_distributed( - const tile_handle &tile, std::size_t N_row, std::size_t N_col, const tile_handle &src) -{ - return hpx::dataflow( - hpx::launch::async, - [=](hpx::future> &&tiled) - { return tile.set_async(cpu::gen_tile_transpose(N_row, N_col, tiled.get())); }, - src.get_async()); -} - -hpx::future> gen_tile_output_distributed( - const tile_handle &tile, std::size_t row, std::size_t N, const std::vector &output); - -HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_output_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_output, gen_tile_output_distributed_action, "gen_tile_output"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_output); - -hpx::future> gen_tile_output_distributed( - const tile_handle &tile, std::size_t row, std::size_t N, const std::vector &output) -{ - return tile.set_async(cpu::gen_tile_output(row, N, output)); -} - -hpx::future> gen_tile_grad_l_distributed( - const tile_handle &tile, - std::size_t N, - const SEKParams &sek_params, - const const_tile_data &distance); -HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_grad_l_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_grad_l, gen_tile_grad_l_distributed_action, "gen_tile_grad_l"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_grad_l); - -hpx::future> gen_tile_grad_l_distributed( - const tile_handle &tile, - std::size_t N, - const SEKParams &sek_params, - const const_tile_data &distance) -{ - return tile.set_async(cpu::gen_tile_grad_l(N, sek_params, distance)); -} - -hpx::future> gen_tile_grad_v_distributed( - const tile_handle &tile, - std::size_t N, - const SEKParams &sek_params, - const const_tile_data &distance); -HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_grad_v_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_grad_v, gen_tile_grad_v_distributed_action, "gen_tile_grad_v"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_grad_l); - -hpx::future> gen_tile_grad_v_distributed( - const tile_handle &tile, - std::size_t N, - const SEKParams &sek_params, - const const_tile_data &distance) -{ - return tile.set_async(cpu::gen_tile_grad_v(N, sek_params, distance)); -} - -hpx::future> gen_tile_zeros_distributed(const tile_handle &tile, std::size_t N); - -HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_zeros_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_zeros, gen_tile_zeros_distributed_action, "gen_tile_output"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_zeros); - -hpx::future> gen_tile_zeros_distributed(const tile_handle &tile, std::size_t N) -{ - return tile.set_async(cpu::gen_tile_zeros(N)); -} - -hpx::future> gen_tile_identity_distributed(const tile_handle &tile, std::size_t N); - -HPX_DEFINE_PLAIN_DIRECT_ACTION(gen_tile_identity_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::gen_tile_identity, gen_tile_identity_distributed_action, "gen_tile_identity"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::gen_tile_identity); - -hpx::future> gen_tile_identity_distributed(const tile_handle &tile, std::size_t N) -{ - return tile.set_async(cpu::gen_tile_identity(N)); -} - -hpx::future> get_matrix_diagonal_distributed(const tile_handle &A, std::size_t M); - -HPX_DEFINE_PLAIN_DIRECT_ACTION(get_matrix_diagonal_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::get_matrix_diagonal, - get_matrix_diagonal_distributed_action, - "get_matrix_diagonal"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::get_matrix_diagonal); - -hpx::future> get_matrix_diagonal_distributed(const tile_handle &A, std::size_t M) -{ - return hpx::dataflow( - hpx::launch::async, - [A, M](hpx::future> &&Ad) - { return A.set_async(cpu::get_matrix_diagonal(Ad.get(), M)); }, - A.get_async()); -} - -hpx::future compute_loss_distributed(const tile_handle &K_diag_tile, - const tile_handle &alpha_tile, - const tile_handle &y_tile, - std::size_t N); - -HPX_DEFINE_PLAIN_DIRECT_ACTION(compute_loss_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::compute_loss, compute_loss_distributed_action, "compute_loss"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::get_matrix_diagonal); - -hpx::future compute_loss_distributed(const tile_handle &K_diag_tile, - const tile_handle &alpha_tile, - const tile_handle &y_tile, - std::size_t N) -{ - return hpx::dataflow( - hpx::launch::async, - [=](hpx::future> &&K_diag_tiled, - hpx::future> &&alpha_tiled, - hpx::future> &&y_tiled) - { return cpu::compute_loss(K_diag_tiled.get(), alpha_tiled.get(), y_tiled.get(), N); }, - K_diag_tile.get_async(), - alpha_tile.get_async(), - y_tile.get_async()); -} - -hpx::future compute_trace_distributed(const tile_handle &diagonal, double trace); - -HPX_DEFINE_PLAIN_DIRECT_ACTION(compute_trace_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::compute_trace, compute_trace_distributed_action, "compute_loss"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::compute_trace); - -hpx::future compute_trace_distributed(const tile_handle &diagonal, double trace) -{ - return hpx::dataflow( - hpx::launch::async, - [=](hpx::future> &&diagonald) { return cpu::compute_trace(diagonald.get(), trace); }, - diagonal.get_async()); -} - -hpx::future -compute_dot_distributed(const tile_handle &vector_T, const tile_handle &vector, double result); - -HPX_DEFINE_PLAIN_DIRECT_ACTION(compute_dot_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::compute_dot, compute_dot_distributed_action, "compute_loss"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::compute_dot); - -hpx::future -compute_dot_distributed(const tile_handle &vector_T, const tile_handle &vector, double result) -{ - return hpx::dataflow( - hpx::launch::async, - [=](hpx::future> &&vector_Td, hpx::future> &&vectord) - { return cpu::compute_dot(vector_Td.get(), vectord.get(), result); }, - vector_T.get_async(), - vector.get_async()); -} - -hpx::future compute_trace_diag_distributed(const tile_handle &tile, double trace, std::size_t N); - -HPX_DEFINE_PLAIN_DIRECT_ACTION(compute_trace_diag_distributed); -GPRAT_DECLARE_PLAIN_ACTION_FOR(&cpu::compute_trace_diag, compute_trace_diag_distributed_action, "compute_loss"); -GPRAT_DEFINE_PLAIN_ACTION_FOR(&cpu::compute_trace_diag); - -hpx::future compute_trace_diag_distributed(const tile_handle &tile, double trace, std::size_t N) -{ - return hpx::dataflow( - hpx::launch::async, - [=](hpx::future> &&tiled) { return cpu::compute_trace_diag(tiled.get(), trace, N); }, - tile.get_async()); -} - gprat_results load_test_data_results(const std::string &filename) { std::ifstream ifs(filename); @@ -639,11 +308,7 @@ void startup() static struct once_dummy_struct { - once_dummy_struct() - { - register_performance_counters(); - register_distributed_tile_counters(); - } + once_dummy_struct() { register_performance_counters(); } } once_dummy; } @@ -657,8 +322,6 @@ bool check_startup(hpx::startup_function_type &startup_func, bool &pre_startup) GPRAT_NS_END -HPX_REGISTER_ACTION(GPRAT_NS::gen_tile_covariance_distributed_action); - HPX_REGISTER_STARTUP_MODULE(GPRAT_NS::check_startup) int hpx_main(hpx::program_options::variables_map &vm) diff --git a/examples/distributed/src/scheduling.cpp b/examples/distributed/src/scheduling.cpp deleted file mode 100644 index 6e1d0489..00000000 --- a/examples/distributed/src/scheduling.cpp +++ /dev/null @@ -1,2 +0,0 @@ -#include "scheduling.hpp" -#include "distributed_tile.hpp" diff --git a/examples/distributed/src/scheduling.hpp b/examples/distributed/src/scheduling.hpp deleted file mode 100644 index a735b06d..00000000 --- a/examples/distributed/src/scheduling.hpp +++ /dev/null @@ -1,10 +0,0 @@ -#pragma once - -#include "gprat/detail/config.hpp" - -#include "gprat/detail/async_helpers.hpp" -#include "gprat/detail/actions.hpp" - -GPRAT_NS_BEGIN - -GPRAT_NS_END From 6fa965f3a3c105fac0c8fe3b54d4c64dcafa4893 Mon Sep 17 00:00:00 2001 From: Tim Niederhausen Date: Tue, 2 Dec 2025 02:07:31 +0100 Subject: [PATCH 56/56] refactor!(core): Use ADL lookup for scheduler customization points A bit more robust --- core/include/gprat/cpu/gp_functions.hpp | 138 +++++++++----------- core/include/gprat/cpu/tiled_algorithms.hpp | 52 ++++---- core/include/gprat/scheduler.hpp | 125 +++++++++++------- core/include/gprat/scheduler/cyclic.hpp | 46 +++---- core/include/gprat/scheduler/sma.hpp | 61 ++++----- test/src/output_correctness.cpp | 2 +- 6 files changed, 213 insertions(+), 211 deletions(-) diff --git a/core/include/gprat/cpu/gp_functions.hpp b/core/include/gprat/cpu/gp_functions.hpp index 55a9e0e3..58a62619 100644 --- a/core/include/gprat/cpu/gp_functions.hpp +++ b/core/include/gprat/cpu/gp_functions.hpp @@ -44,7 +44,7 @@ cholesky(Scheduler &sched, sched, n_tiles * n_tiles, [&](std::size_t tile_index) - { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + { return covariance_tile_on(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); for (std::size_t row = 0; row < n_tiles; row++) { @@ -52,7 +52,7 @@ cholesky(Scheduler &sched, { K_tiles[row * n_tiles + col] = detail::named_make_tile( sched, - schedule::covariance_tile(sched, n_tiles, row, col), + covariance_tile_on(sched, n_tiles, row, col), "assemble_tiled_K", K_tiles[row * n_tiles + col], row, @@ -132,7 +132,7 @@ predict(Scheduler &sched, sched, n_tiles * n_tiles, [&](std::size_t tile_index) - { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + { return covariance_tile_on(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); for (std::size_t row = 0; row < n_tiles; row++) { @@ -140,7 +140,7 @@ predict(Scheduler &sched, { K_tiles[row * n_tiles + col] = detail::named_make_tile( sched, - schedule::covariance_tile(sched, n_tiles, row, col), + covariance_tile_on(sched, n_tiles, row, col), "assemble_tiled_K", K_tiles[row * n_tiles + col], row, @@ -163,19 +163,19 @@ predict(Scheduler &sched, sched, m_tiles * n_tiles, [&](std::size_t tile_index) - { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + { return covariance_tile_on(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); // Tiled solution auto prediction_tiles = make_tiled_dataset( - sched, m_tiles, [&](std::size_t tile_index) { return schedule::prediction_tile(sched, m_tiles, tile_index); }); + sched, m_tiles, [&](std::size_t tile_index) { return prediction_tile_on(sched, m_tiles, tile_index); }); // Tiled intermediate solution auto alpha_tiles = make_tiled_dataset( - sched, n_tiles, [&](std::size_t tile_index) { return schedule::alpha_tile(sched, n_tiles, tile_index); }); + sched, n_tiles, [&](std::size_t tile_index) { return alpha_tile_on(sched, n_tiles, tile_index); }); for (std::size_t i = 0; i < n_tiles; i++) { alpha_tiles[i] = detail::named_make_tile( sched, - schedule::alpha_tile(sched, n_tiles, i), + alpha_tile_on(sched, n_tiles, i), "assemble_tiled_alpha", alpha_tiles[i], i, @@ -189,7 +189,7 @@ predict(Scheduler &sched, { cross_covariance_tiles[i * n_tiles + j] = detail::named_make_tile( sched, - schedule::cross_covariance_tile(sched, n_tiles, i, j), + cross_covariance_tile_on(sched, n_tiles, i, j), "assemble_pred", cross_covariance_tiles[i * n_tiles + j], i, @@ -206,7 +206,7 @@ predict(Scheduler &sched, for (std::size_t i = 0; i < m_tiles; i++) { prediction_tiles[i] = detail::named_make_tile( - sched, schedule::prediction_tile(sched, m_tiles, i), "assemble_tiled", prediction_tiles[i], m_tile_size); + sched, prediction_tile_on(sched, m_tiles, i), "assemble_tiled", prediction_tiles[i], m_tile_size); } // Launch asynchronous triangular solve L * (L^T * alpha) = y @@ -289,7 +289,7 @@ std::vector> predict_with_uncertainty( sched, n_tiles * n_tiles, [&](std::size_t tile_index) - { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + { return covariance_tile_on(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); for (std::size_t row = 0; row < n_tiles; row++) { @@ -297,7 +297,7 @@ std::vector> predict_with_uncertainty( { K_tiles[row * n_tiles + col] = detail::named_make_tile( sched, - schedule::covariance_tile(sched, n_tiles, row, col), + covariance_tile_on(sched, n_tiles, row, col), "assemble_tiled_K", K_tiles[row * n_tiles + col], row, @@ -317,12 +317,12 @@ std::vector> predict_with_uncertainty( // Tiled intermediate solution auto alpha_tiles = make_tiled_dataset( - sched, n_tiles, [&](std::size_t tile_index) { return schedule::alpha_tile(sched, n_tiles, tile_index); }); + sched, n_tiles, [&](std::size_t tile_index) { return alpha_tile_on(sched, n_tiles, tile_index); }); for (std::size_t i = 0; i < n_tiles; i++) { alpha_tiles[i] = detail::named_make_tile( sched, - schedule::alpha_tile(sched, n_tiles, i), + alpha_tile_on(sched, n_tiles, i), "assemble_tiled_alpha", alpha_tiles[i], i, @@ -335,14 +335,14 @@ std::vector> predict_with_uncertainty( sched, m_tiles * n_tiles, [&](std::size_t tile_index) - { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + { return covariance_tile_on(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); for (std::size_t i = 0; i < m_tiles; i++) { for (std::size_t j = 0; j < n_tiles; j++) { cross_covariance_tiles[i * n_tiles + j] = detail::named_make_tile( sched, - schedule::cross_covariance_tile(sched, n_tiles, i, j), + cross_covariance_tile_on(sched, n_tiles, i, j), "assemble_pred", cross_covariance_tiles[i * n_tiles + j], i, @@ -358,11 +358,11 @@ std::vector> predict_with_uncertainty( // Tiled solution auto prediction_tiles = make_tiled_dataset( - sched, m_tiles, [&](std::size_t tile_index) { return schedule::prediction_tile(sched, m_tiles, tile_index); }); + sched, m_tiles, [&](std::size_t tile_index) { return prediction_tile_on(sched, m_tiles, tile_index); }); for (std::size_t i = 0; i < m_tiles; i++) { prediction_tiles[i] = detail::named_make_tile( - sched, schedule::prediction_tile(sched, m_tiles, i), "assemble_tiled", prediction_tiles[i], m_tile_size); + sched, prediction_tile_on(sched, m_tiles, i), "assemble_tiled", prediction_tiles[i], m_tile_size); } // Launch asynchronous triangular solve L * (L^T * alpha) = y @@ -381,14 +381,14 @@ std::vector> predict_with_uncertainty( sched, n_tiles * m_tiles, [&](std::size_t tile_index) - { return schedule::t_cross_covariance_tile(sched, m_tiles, tile_index / m_tiles, tile_index % m_tiles); }); + { return t_cross_covariance_tile_on(sched, m_tiles, tile_index / m_tiles, tile_index % m_tiles); }); for (std::size_t j = 0; j < n_tiles; j++) { for (std::size_t i = 0; i < m_tiles; i++) { t_cross_covariance_tiles[j * m_tiles + i] = detail::named_make_tile( sched, - schedule::t_cross_covariance_tile(sched, m_tiles, j, i), + t_cross_covariance_tile_on(sched, m_tiles, j, i), "assemble_pred", t_cross_covariance_tiles[j * m_tiles + i], m_tile_size, @@ -399,12 +399,12 @@ std::vector> predict_with_uncertainty( // Tiled prior covariance matrix diagonal diag(K_MxM) auto prior_K_tiles = make_tiled_dataset( - sched, m_tiles, [&](std::size_t tile_index) { return schedule::prior_K_tile(sched, n_tiles, 0, tile_index); }); + sched, m_tiles, [&](std::size_t tile_index) { return prior_K_tile_on(sched, n_tiles, 0, tile_index); }); for (std::size_t i = 0; i < m_tiles; i++) { prior_K_tiles[i] = detail::named_make_tile( sched, - schedule::prior_K_tile(sched, m_tiles, 0, i), + prior_K_tile_on(sched, m_tiles, 0, i), "assemble_tiled", prior_K_tiles[i], i, @@ -417,15 +417,11 @@ std::vector> predict_with_uncertainty( // Tiled uncertainty solution auto uncertainty_tiles = make_tiled_dataset( - sched, m_tiles, [&](std::size_t tile_index) { return schedule::uncertainty_tile(sched, m_tiles, tile_index); }); + sched, m_tiles, [&](std::size_t tile_index) { return uncertainty_tile_on(sched, m_tiles, tile_index); }); for (std::size_t i = 0; i < m_tiles; i++) { uncertainty_tiles[i] = detail::named_make_tile( - sched, - schedule::uncertainty_tile(sched, m_tiles, i), - "assemble_prior_inter", - uncertainty_tiles[i], - m_tile_size); + sched, uncertainty_tile_on(sched, m_tiles, i), "assemble_prior_inter", uncertainty_tiles[i], m_tile_size); } // Launch asynchronous triangular solve L * V = cross(K)^T @@ -522,14 +518,14 @@ std::vector> predict_with_full_cov( sched, n_tiles * n_tiles, [&](std::size_t tile_index) - { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + { return covariance_tile_on(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); for (std::size_t row = 0; row < n_tiles; row++) { for (std::size_t col = 0; col <= row; col++) { K_tiles[row * n_tiles + col] = detail::named_make_tile( sched, - schedule::covariance_tile(sched, n_tiles, row, col), + covariance_tile_on(sched, n_tiles, row, col), "assemble_tiled_K", K_tiles[row * n_tiles + col], row, @@ -549,12 +545,12 @@ std::vector> predict_with_full_cov( // Tiled intermediate solution auto alpha_tiles = make_tiled_dataset( - sched, n_tiles, [&](std::size_t tile_index) { return schedule::alpha_tile(sched, n_tiles, tile_index); }); + sched, n_tiles, [&](std::size_t tile_index) { return alpha_tile_on(sched, n_tiles, tile_index); }); for (std::size_t i = 0; i < n_tiles; i++) { alpha_tiles[i] = detail::named_make_tile( sched, - schedule::alpha_tile(sched, n_tiles, i), + alpha_tile_on(sched, n_tiles, i), "assemble_tiled_alpha", alpha_tiles[i], i, @@ -567,14 +563,14 @@ std::vector> predict_with_full_cov( sched, m_tiles * n_tiles, [&](std::size_t tile_index) - { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + { return covariance_tile_on(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); for (std::size_t i = 0; i < m_tiles; i++) { for (std::size_t j = 0; j < n_tiles; j++) { cross_covariance_tiles[i * n_tiles + j] = detail::named_make_tile( sched, - schedule::cross_covariance_tile(sched, n_tiles, i, j), + cross_covariance_tile_on(sched, n_tiles, i, j), "assemble_pred", cross_covariance_tiles[i * n_tiles + j], i, @@ -590,11 +586,11 @@ std::vector> predict_with_full_cov( // Tiled solution auto prediction_tiles = make_tiled_dataset( - sched, m_tiles, [&](std::size_t tile_index) { return schedule::prediction_tile(sched, n_tiles, tile_index); }); + sched, m_tiles, [&](std::size_t tile_index) { return prediction_tile_on(sched, n_tiles, tile_index); }); for (std::size_t i = 0; i < m_tiles; i++) { prediction_tiles[i] = detail::named_make_tile( - sched, schedule::prediction_tile(sched, m_tiles, i), "assemble_tiled", prediction_tiles[i], m_tile_size); + sched, prediction_tile_on(sched, m_tiles, i), "assemble_tiled", prediction_tiles[i], m_tile_size); } // Launch asynchronous triangular solve L * (L^T * alpha) = y @@ -613,14 +609,14 @@ std::vector> predict_with_full_cov( sched, n_tiles * m_tiles, [&](std::size_t tile_index) - { return schedule::t_cross_covariance_tile(sched, m_tiles, tile_index / m_tiles, tile_index % m_tiles); }); + { return t_cross_covariance_tile_on(sched, m_tiles, tile_index / m_tiles, tile_index % m_tiles); }); for (std::size_t j = 0; j < n_tiles; j++) { for (std::size_t i = 0; i < m_tiles; i++) { t_cross_covariance_tiles[j * m_tiles + i] = detail::named_make_tile( sched, - schedule::t_cross_covariance_tile(sched, m_tiles, j, i), + t_cross_covariance_tile_on(sched, m_tiles, j, i), "assemble_pred", t_cross_covariance_tiles[j * m_tiles + i], m_tile_size, @@ -634,14 +630,14 @@ std::vector> predict_with_full_cov( sched, m_tiles * m_tiles, [&](std::size_t tile_index) - { return schedule::prior_K_tile(sched, n_tiles, tile_index / m_tiles, tile_index % m_tiles); }); + { return prior_K_tile_on(sched, n_tiles, tile_index / m_tiles, tile_index % m_tiles); }); for (std::size_t i = 0; i < m_tiles; i++) { for (std::size_t j = 0; j <= i; j++) { prior_K_tiles[i * m_tiles + j] = detail::named_make_tile( sched, - schedule::prior_K_tile(sched, m_tiles, i, j), + prior_K_tile_on(sched, m_tiles, i, j), "assemble_prior_tiled", prior_K_tiles[i * m_tiles + j], i, @@ -655,7 +651,7 @@ std::vector> predict_with_full_cov( { prior_K_tiles[j * m_tiles + i] = detail::named_make_tile( sched, - schedule::prior_K_tile(sched, m_tiles, j, i), + prior_K_tile_on(sched, m_tiles, j, i), "assemble_prior_tiled", prior_K_tiles[j * m_tiles + i], m_tile_size, @@ -667,15 +663,11 @@ std::vector> predict_with_full_cov( // Tiled uncertainty solution auto uncertainty_tiles = make_tiled_dataset( - sched, m_tiles, [&](std::size_t tile_index) { return schedule::uncertainty_tile(sched, m_tiles, tile_index); }); + sched, m_tiles, [&](std::size_t tile_index) { return uncertainty_tile_on(sched, m_tiles, tile_index); }); for (std::size_t i = 0; i < m_tiles; i++) { uncertainty_tiles[i] = detail::named_make_tile( - sched, - schedule::uncertainty_tile(sched, m_tiles, i), - "assemble_prior_inter", - uncertainty_tiles[i], - m_tile_size); + sched, uncertainty_tile_on(sched, m_tiles, i), "assemble_prior_inter", uncertainty_tiles[i], m_tile_size); } // Launch asynchronous triangular solve L * V = cross(K)^T @@ -762,14 +754,14 @@ double calculate_loss(Scheduler &sched, sched, n_tiles * n_tiles, [&](std::size_t tile_index) - { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + { return covariance_tile_on(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); for (std::size_t row = 0; row < n_tiles; row++) { for (std::size_t col = 0; col <= row; col++) { K_tiles[row * n_tiles + col] = detail::named_make_tile( sched, - schedule::covariance_tile(sched, n_tiles, row, col), + covariance_tile_on(sched, n_tiles, row, col), "assemble_tiled_K", K_tiles[row * n_tiles + col], row, @@ -783,12 +775,12 @@ double calculate_loss(Scheduler &sched, // Tiled intermediate solution auto alpha_tiles = make_tiled_dataset( - sched, n_tiles, [&](std::size_t tile_index) { return schedule::alpha_tile(sched, n_tiles, tile_index); }); + sched, n_tiles, [&](std::size_t tile_index) { return alpha_tile_on(sched, n_tiles, tile_index); }); for (std::size_t i = 0; i < n_tiles; i++) { alpha_tiles[i] = detail::named_make_tile( sched, - schedule::alpha_tile(sched, n_tiles, i), + alpha_tile_on(sched, n_tiles, i), "assemble_tiled_alpha", alpha_tiles[i], i, @@ -798,12 +790,12 @@ double calculate_loss(Scheduler &sched, // Tiled output auto y_tiles = make_tiled_dataset( - sched, n_tiles, [&](std::size_t tile_index) { return schedule::prediction_tile(sched, n_tiles, tile_index); }); + sched, n_tiles, [&](std::size_t tile_index) { return prediction_tile_on(sched, n_tiles, tile_index); }); for (std::size_t i = 0; i < n_tiles; i++) { y_tiles[i] = detail::named_make_tile( sched, - schedule::prediction_tile(sched, n_tiles, i), + prediction_tile_on(sched, n_tiles, i), "assemble_tiled_alpha", y_tiles[i], i, @@ -890,18 +882,12 @@ optimize(Scheduler &sched, // Tiled output auto y_tiles = make_tiled_dataset( - sched, n_tiles, [&](std::size_t tile_index) { return schedule::prediction_tile(sched, n_tiles, tile_index); }); + sched, n_tiles, [&](std::size_t tile_index) { return prediction_tile_on(sched, n_tiles, tile_index); }); // Launch asynchronous assembly of output y for (std::size_t i = 0; i < n_tiles; i++) { y_tiles[i] = detail::named_make_tile( - sched, - schedule::prediction_tile(sched, n_tiles, i), - "assemble_y", - y_tiles[i], - i, - n_tile_size, - training_output); + sched, prediction_tile_on(sched, n_tiles, i), "assemble_y", y_tiles[i], i, n_tile_size, training_output); } ////////////////////////////////////////////////////////////////////////////// @@ -912,18 +898,18 @@ optimize(Scheduler &sched, sched, n_tiles * n_tiles, [&](std::size_t tile_index) - { return schedule::covariance_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + { return covariance_tile_on(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); // Tiled inverse covariance matrix K^-1_NxN auto K_inv_tiles = make_tiled_dataset( sched, n_tiles * n_tiles, [&](std::size_t tile_index) - { return schedule::K_inv_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + { return K_inv_tile_on(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); // Tiled intermediate solution auto alpha_tiles = make_tiled_dataset( - sched, n_tiles, [&](std::size_t tile_index) { return schedule::alpha_tile(sched, n_tiles, tile_index); }); + sched, n_tiles, [&](std::size_t tile_index) { return alpha_tile_on(sched, n_tiles, tile_index); }); // Tiled future data structures for gradients @@ -932,20 +918,20 @@ optimize(Scheduler &sched, sched, n_tiles * n_tiles, [&](std::size_t tile_index) - { return schedule::K_grad_v_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + { return K_grad_v_tile_on(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); // Tiled covariance with gradient l auto grad_l_tiles = make_tiled_dataset( sched, n_tiles * n_tiles, [&](std::size_t tile_index) - { return schedule::K_grad_l_tile(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); + { return K_grad_l_tile_on(sched, n_tiles, tile_index / n_tiles, tile_index % n_tiles); }); auto inter_alpha = make_tiled_dataset( - sched, n_tiles, [&](std::size_t tile_index) { return schedule::inter_alpha_tile(sched, n_tiles, tile_index); }); + sched, n_tiles, [&](std::size_t tile_index) { return inter_alpha_tile_on(sched, n_tiles, tile_index); }); auto diag_tiles = make_tiled_dataset( - sched, n_tiles, [&](std::size_t tile_index) { return schedule::diag_tile(sched, n_tiles, tile_index); }); + sched, n_tiles, [&](std::size_t tile_index) { return diag_tile_on(sched, n_tiles, tile_index); }); ////////////////////////////////////////////////////////////////////////////// // Perform optimization @@ -965,7 +951,7 @@ optimize(Scheduler &sched, K_tiles[i * n_tiles + j] = detail::named_make_tile( sched, - schedule::covariance_tile(sched, n_tiles, i, j), + covariance_tile_on(sched, n_tiles, i, j), "assemble_K", K_tiles[i * n_tiles + j], i, @@ -977,7 +963,7 @@ optimize(Scheduler &sched, { grad_l_tiles[i * n_tiles + j] = detail::named_make_tile( sched, - schedule::K_grad_l_tile(sched, n_tiles, i, j), + K_grad_l_tile_on(sched, n_tiles, i, j), "assemble_gradl", grad_l_tiles[i * n_tiles + j], n_tile_size, @@ -987,7 +973,7 @@ optimize(Scheduler &sched, { grad_l_tiles[j * n_tiles + i] = detail::named_make_tile( sched, - schedule::K_grad_l_tile(sched, n_tiles, j, i), + K_grad_l_tile_on(sched, n_tiles, j, i), "assemble_gradl_t", grad_l_tiles[j * n_tiles + i], n_tile_size, @@ -1000,7 +986,7 @@ optimize(Scheduler &sched, { grad_v_tiles[i * n_tiles + j] = detail::named_make_tile( sched, - schedule::K_grad_v_tile(sched, n_tiles, i, j), + K_grad_v_tile_on(sched, n_tiles, i, j), "assemble_gradv", grad_v_tiles[i * n_tiles + j], n_tile_size, @@ -1010,7 +996,7 @@ optimize(Scheduler &sched, { grad_v_tiles[j * n_tiles + i] = detail::named_make_tile( sched, - schedule::K_grad_v_tile(sched, n_tiles, j, i), + K_grad_v_tile_on(sched, n_tiles, j, i), "assemble_gradv_t", grad_v_tiles[j * n_tiles + i], n_tile_size, @@ -1025,7 +1011,7 @@ optimize(Scheduler &sched, for (std::size_t i = 0; i < n_tiles; i++) { alpha_tiles[i] = detail::named_make_tile( - sched, schedule::alpha_tile(sched, n_tiles, i), "assemble_tiled_alpha", alpha_tiles[i], n_tile_size); + sched, alpha_tile_on(sched, n_tiles, i), "assemble_tiled_alpha", alpha_tiles[i], n_tile_size); } for (std::size_t i = 0; i < n_tiles; i++) @@ -1036,7 +1022,7 @@ optimize(Scheduler &sched, { K_inv_tiles[i * n_tiles + j] = detail::named_make_tile( sched, - schedule::K_inv_tile(sched, n_tiles, i, j), + K_inv_tile_on(sched, n_tiles, i, j), "assemble_identity_matrix", K_inv_tiles[i * n_tiles + j], n_tile_size); @@ -1045,7 +1031,7 @@ optimize(Scheduler &sched, { K_inv_tiles[i * n_tiles + j] = detail::named_make_tile( sched, - schedule::K_inv_tile(sched, n_tiles, i, j), + K_inv_tile_on(sched, n_tiles, i, j), "assemble_identity_matrix", K_inv_tiles[i * n_tiles + j], n_tile_size * n_tile_size); diff --git a/core/include/gprat/cpu/tiled_algorithms.hpp b/core/include/gprat/cpu/tiled_algorithms.hpp index 718e4d5b..2438fab6 100644 --- a/core/include/gprat/cpu/tiled_algorithms.hpp +++ b/core/include/gprat/cpu/tiled_algorithms.hpp @@ -52,13 +52,13 @@ void right_looking_cholesky_tiled(Scheduler &sched, Tiles &tiles, std::size_t N, { // POTRF: Compute Cholesky factor L tiles[k * n_tiles + k] = detail::named_dataflow( - sched, schedule::cholesky_potrf(sched, n_tiles, k), "cholesky_tiled", tiles[k * n_tiles + k], N); + sched, cholesky_potrf_on(sched, n_tiles, k), "cholesky_tiled", tiles[k * n_tiles + k], N); for (std::size_t m = k + 1; m < n_tiles; m++) { // TRSM: Solve X * L^T = A tiles[m * n_tiles + k] = detail::named_dataflow( sched, - schedule::cholesky_trsm(sched, n_tiles, k, m), + cholesky_trsm_on(sched, n_tiles, k, m), "cholesky_tiled", tiles[k * n_tiles + k], tiles[m * n_tiles + k], @@ -72,7 +72,7 @@ void right_looking_cholesky_tiled(Scheduler &sched, Tiles &tiles, std::size_t N, // SYRK: A = A - B * B^T tiles[m * n_tiles + m] = detail::named_dataflow( sched, - schedule::cholesky_syrk(sched, n_tiles, m), + cholesky_syrk_on(sched, n_tiles, m), "cholesky_tiled", tiles[m * n_tiles + m], tiles[m * n_tiles + k], @@ -82,7 +82,7 @@ void right_looking_cholesky_tiled(Scheduler &sched, Tiles &tiles, std::size_t N, // GEMM: C = C - A * B^T tiles[m * n_tiles + n] = detail::named_dataflow( sched, - schedule::cholesky_gemm(sched, n_tiles, k, m, n), + cholesky_gemm_on(sched, n_tiles, k, m, n), "cholesky_tiled", tiles[m * n_tiles + k], tiles[n * n_tiles + k], @@ -115,7 +115,7 @@ void forward_solve_tiled(Scheduler &sched, Tiles &ft_tiles, Tiles &ft_rhs, std:: // TRSM: Solve L * x = a ft_rhs[k] = detail::named_dataflow( sched, - schedule::solve_trsv(sched, n_tiles, k), + solve_trsv_on(sched, n_tiles, k), "triangular_solve_tiled", ft_tiles[k * n_tiles + k], ft_rhs[k], @@ -126,7 +126,7 @@ void forward_solve_tiled(Scheduler &sched, Tiles &ft_tiles, Tiles &ft_rhs, std:: // GEMV: b = b - A * a ft_rhs[m] = detail::named_dataflow( sched, - schedule::solve_gemv(sched, n_tiles, k, m), + solve_gemv_on(sched, n_tiles, k, m), "triangular_solve_tiled", ft_tiles[m * n_tiles + k], ft_rhs[k], @@ -156,7 +156,7 @@ void backward_solve_tiled(Scheduler &sched, Tiles &ft_tiles, Tiles &ft_rhs, std: // TRSM: Solve L^T * x = a ft_rhs[k] = detail::named_dataflow( sched, - schedule::solve_trsm(sched, n_tiles, k), + solve_trsm_on(sched, n_tiles, k), "triangular_solve_tiled", ft_tiles[k * n_tiles + k], ft_rhs[k], @@ -168,7 +168,7 @@ void backward_solve_tiled(Scheduler &sched, Tiles &ft_tiles, Tiles &ft_rhs, std: // GEMV:b = b - A^T * a ft_rhs[m] = detail::named_dataflow( sched, - schedule::solve_gemv(sched, n_tiles, k, m), + solve_gemv_on(sched, n_tiles, k, m), "triangular_solve_tiled", ft_tiles[k * n_tiles + m], ft_rhs[k], @@ -208,7 +208,7 @@ void forward_solve_tiled_matrix( // TRSM: solve L * X = A ft_rhs[k * m_tiles + c] = detail::named_dataflow( sched, - schedule::solve_matrix_trsm(sched, m_tiles, c, k), + solve_matrix_trsm_on(sched, m_tiles, c, k), "triangular_solve_tiled_matrix", ft_tiles[k * n_tiles + k], ft_rhs[k * m_tiles + c], @@ -221,7 +221,7 @@ void forward_solve_tiled_matrix( // GEMM: C = C - A * B ft_rhs[m * m_tiles + c] = detail::named_dataflow( sched, - schedule::solve_matrix_gemm(sched, m_tiles, c, k, m), + solve_matrix_gemm_on(sched, m_tiles, c, k, m), "triangular_solve_tiled_matrix", ft_tiles[m * n_tiles + k], ft_rhs[k * m_tiles + c], @@ -264,7 +264,7 @@ void backward_solve_tiled_matrix( // TRSM: solve L^T * X = A ft_rhs[k * m_tiles + c] = detail::named_dataflow( sched, - schedule::solve_matrix_trsm(sched, m_tiles, c, k), + solve_matrix_trsm_on(sched, m_tiles, c, k), "triangular_solve_tiled_matrix", ft_tiles[k * n_tiles + k], ft_rhs[k * m_tiles + c], @@ -278,7 +278,7 @@ void backward_solve_tiled_matrix( // GEMM: C = C - A^T * B ft_rhs[m * m_tiles + c] = detail::named_dataflow( sched, - schedule::solve_matrix_gemm(sched, m_tiles, c, k, m), + solve_matrix_gemm_on(sched, m_tiles, c, k, m), "triangular_solve_tiled_matrix", ft_tiles[k * n_tiles + m], ft_rhs[k * m_tiles + c], @@ -320,7 +320,7 @@ void matrix_vector_tiled(Scheduler &sched, { ft_rhs[k] = detail::named_dataflow( sched, - schedule::multiply_gemv(sched, n_tiles, k, m), + multiply_gemv_on(sched, n_tiles, k, m), "prediction_tiled", ft_tiles[k * n_tiles + m], ft_vector[m], @@ -361,7 +361,7 @@ void symmetric_matrix_matrix_diagonal_tiled( // V^T * V <=> cross(K) * K^-1 * cross(K)^T ft_vector[i] = detail::named_dataflow( sched, - schedule::k_rank_dot_diag_syrk(sched, m_tiles, i), + k_rank_dot_diag_syrk_on(sched, m_tiles, i), "posterior_tiled", ft_tiles[n * m_tiles + i], ft_vector[i], @@ -401,7 +401,7 @@ void symmetric_matrix_matrix_tiled( // GEMM: C = C - A^T * B ft_result[c * m_tiles + k] = detail::named_dataflow( sched, - schedule::k_rank_gemm(sched, m_tiles, c, k, m), + k_rank_gemm_on(sched, m_tiles, c, k, m), "triangular_solve_tiled_matrix", ft_tiles[m * m_tiles + c], ft_tiles[m * m_tiles + k], @@ -430,7 +430,7 @@ void vector_difference_tiled( for (std::size_t i = 0; i < m_tiles; i++) { ft_subtrahend[i] = detail::named_dataflow( - sched, schedule::vector_axpy(sched, m_tiles, i), "uncertainty_tiled", ft_minuend[i], ft_subtrahend[i], M); + sched, vector_axpy_on(sched, m_tiles, i), "uncertainty_tiled", ft_minuend[i], ft_subtrahend[i], M); } } @@ -447,7 +447,7 @@ void matrix_diagonal_tiled(Scheduler &sched, Tiles &ft_tiles, Tiles &ft_vector, for (std::size_t i = 0; i < m_tiles; i++) { ft_vector[i] = detail::named_dataflow( - sched, schedule::get_diagonal(sched, m_tiles, i), "uncertainty_tiled", ft_tiles[i * m_tiles + i], M); + sched, get_diagonal_on(sched, m_tiles, i), "uncertainty_tiled", ft_tiles[i * m_tiles + i], M); } } @@ -473,7 +473,7 @@ compute_loss_tiled(Scheduler &sched, Tiles &ft_tiles, Tiles &ft_alpha, Tiles &ft { loss_tiled.push_back(detail::named_dataflow( sched, - schedule::compute_loss(sched, n_tiles, k), + compute_loss_on(sched, n_tiles, k), "loss_tiled", ft_tiles[k * n_tiles + k], ft_alpha[k], @@ -535,9 +535,9 @@ void update_hyperparameter_tiled_lengthscale( for (std::size_t d = 0; d < n_tiles; d++) { diag_tiles[d] = detail::named_make_tile( - sched, schedule::diag_tile(sched, n_tiles, d), "assemble", diag_tiles[d], N); + sched, diag_tile_on(sched, n_tiles, d), "assemble", diag_tiles[d], N); inter_alpha[d] = detail::named_make_tile( - sched, schedule::inter_alpha_tile(sched, n_tiles, d), "assemble", inter_alpha[d], N); + sched, inter_alpha_tile_on(sched, n_tiles, d), "assemble", inter_alpha[d], N); } //////////////////////////////////// @@ -550,7 +550,7 @@ void update_hyperparameter_tiled_lengthscale( { diag_tiles[i] = detail::named_dataflow( sched, - schedule::diag_tile(sched, n_tiles, i), + diag_tile_on(sched, n_tiles, i), "trace", ft_invK[i * n_tiles + j], ft_gradK_param[j * n_tiles + i], @@ -563,7 +563,7 @@ void update_hyperparameter_tiled_lengthscale( for (std::size_t j = 0; j < n_tiles; ++j) { trace = detail::named_dataflow( - sched, schedule::diag_tile(sched, n_tiles, j), "trace", diag_tiles[j], trace); + sched, diag_tile_on(sched, n_tiles, j), "trace", diag_tiles[j], trace); } // Not sure if can be done this way // Step 2: Compute alpha^T * grad(K)_param * alpha (with alpha = inv(K) * y) @@ -574,7 +574,7 @@ void update_hyperparameter_tiled_lengthscale( { inter_alpha[k] = detail::named_dataflow( sched, - schedule::inter_alpha_tile(sched, n_tiles, k), + inter_alpha_tile_on(sched, n_tiles, k), "gemv", ft_gradK_param[k * n_tiles + m], ft_alpha[m], @@ -589,7 +589,7 @@ void update_hyperparameter_tiled_lengthscale( for (std::size_t j = 0; j < n_tiles; ++j) { dot = detail::named_dataflow( - sched, schedule::inter_alpha_tile(sched, n_tiles, j), "grad_right_tiled", inter_alpha[j], ft_alpha[j], dot); + sched, inter_alpha_tile_on(sched, n_tiles, j), "grad_right_tiled", inter_alpha[j], ft_alpha[j], dot); } impl::update_parameters( @@ -634,14 +634,14 @@ void update_hyperparameter_tiled_noise_variance( for (std::size_t j = 0; j < n_tiles; ++j) { trace = detail::named_dataflow( - sched, schedule::K_inv_tile(sched, n_tiles, j, j), "grad_left_tiled", ft_invK[j * n_tiles + j], trace, N); + sched, K_inv_tile_on(sched, n_tiles, j, j), "grad_left_tiled", ft_invK[j * n_tiles + j], trace, N); } //////////////////////////////////// // Step 2: Compute the alpha^T * alpha * noise_variance for (std::size_t j = 0; j < n_tiles; ++j) { dot = detail::named_dataflow( - sched, schedule::alpha_tile(sched, n_tiles, j), "grad_right_tiled", ft_alpha[j], ft_alpha[j], dot); + sched, alpha_tile_on(sched, n_tiles, j), "grad_right_tiled", ft_alpha[j], ft_alpha[j], dot); } factor = compute_sigmoid(to_unconstrained(sek_params.noise_variance, true)); diff --git a/core/include/gprat/scheduler.hpp b/core/include/gprat/scheduler.hpp index 2da7ccd7..80c026ca 100644 --- a/core/include/gprat/scheduler.hpp +++ b/core/include/gprat/scheduler.hpp @@ -33,150 +33,175 @@ tiled_dataset_local make_tiled_dataset(const tiled_scheduler_local &, std::si return std::vector>>{ num_tiles }; } -/// @brief This namespace contains the operation placement functions for all schedulers. -namespace schedule -{ - -#ifdef _MSC_VER -#pragma warning(push) -#pragma warning(disable : 4100) -#endif - // ============================================================= // local scheduler -constexpr std::size_t -covariance_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +constexpr std::size_t covariance_tile_on( + const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*row*/, std::size_t /*col*/) { return 0; } -constexpr std::size_t -cross_covariance_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +constexpr std::size_t cross_covariance_tile_on( + const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*row*/, std::size_t /*col*/) { return 0; } -constexpr std::size_t alpha_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t i) { return 0; } - -constexpr std::size_t prediction_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t i) +constexpr std::size_t alpha_tile_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*i*/) { return 0; } constexpr std::size_t -t_cross_covariance_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +prediction_tile_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*i*/) { return 0; } -constexpr std::size_t -prior_K_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +constexpr std::size_t t_cross_covariance_tile_on( + const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*row*/, std::size_t /*col*/) { return 0; } -constexpr std::size_t -K_inv_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +constexpr std::size_t prior_K_tile_on( + const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*row*/, std::size_t /*col*/) { return 0; } -constexpr std::size_t -K_grad_v_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +constexpr std::size_t K_inv_tile_on( + const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*row*/, std::size_t /*col*/) { return 0; } -constexpr std::size_t -K_grad_l_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t row, std::size_t col) +constexpr std::size_t K_grad_v_tile_on( + const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*row*/, std::size_t /*col*/) { return 0; } -constexpr std::size_t uncertainty_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t i) +constexpr std::size_t K_grad_l_tile_on( + const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*row*/, std::size_t /*col*/) { return 0; } -constexpr std::size_t inter_alpha_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t i) +constexpr std::size_t +uncertainty_tile_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*i*/) { return 0; } -constexpr std::size_t diag_tile(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t i) { return 0; } - -constexpr std::size_t cholesky_potrf(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k) +constexpr std::size_t +inter_alpha_tile_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*i*/) { return 0; } -constexpr std::size_t cholesky_syrk(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t m) +constexpr std::size_t diag_tile_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*i*/) { return 0; } constexpr std::size_t -cholesky_trsm(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k, std::size_t m) +cholesky_potrf_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*k*/) { return 0; } constexpr std::size_t -cholesky_gemm(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k, std::size_t m, std::size_t n) +cholesky_syrk_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*m*/) { return 0; } -constexpr std::size_t solve_trsv(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k) { return 0; } +constexpr std::size_t +cholesky_trsm_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*k*/, std::size_t /*m*/) +{ + return 0; +} -constexpr std::size_t solve_trsm(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k) { return 0; } +constexpr std::size_t cholesky_gemm_on(const tiled_scheduler_local & /*sched*/, + std::size_t /*n_tiles*/, + std::size_t /*k*/, + std::size_t /*m*/, + std::size_t /*n*/) +{ + return 0; +} -constexpr std::size_t solve_gemv(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k, std::size_t m) +constexpr std::size_t solve_trsv_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*k*/) { return 0; } -constexpr std::size_t -solve_matrix_trsm(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t c, std::size_t k) +constexpr std::size_t solve_trsm_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*k*/) { return 0; } constexpr std::size_t -solve_matrix_gemm(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) +solve_gemv_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*k*/, std::size_t /*m*/) { return 0; } -constexpr std::size_t -multiply_gemv(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k, std::size_t m) +constexpr std::size_t solve_matrix_trsm_on( + const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*c*/, std::size_t /*k*/) { return 0; } -constexpr std::size_t k_rank_dot_diag_syrk(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k) +constexpr std::size_t solve_matrix_gemm_on(const tiled_scheduler_local & /*sched*/, + std::size_t /*n_tiles*/, + std::size_t /*c*/, + std::size_t /*k*/, + std::size_t /*m*/) { return 0; } constexpr std::size_t -k_rank_gemm(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) +multiply_gemv_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*k*/, std::size_t /*m*/) { return 0; } -constexpr std::size_t vector_axpy(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k) { return 0; } +constexpr std::size_t +k_rank_dot_diag_syrk_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*k*/) +{ + return 0; +} -constexpr std::size_t get_diagonal(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k) { return 0; } +constexpr std::size_t k_rank_gemm_on(const tiled_scheduler_local & /*sched*/, + std::size_t /*n_tiles*/, + std::size_t /*c*/, + std::size_t /*k*/, + std::size_t /*m*/) +{ + return 0; +} -constexpr std::size_t compute_loss(const tiled_scheduler_local &sched, std::size_t n_tiles, std::size_t k) { return 0; } +constexpr std::size_t +vector_axpy_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*k*/) +{ + return 0; +} -#ifdef _MSC_VER -#pragma warning(pop) -#endif +constexpr std::size_t +get_diagonal_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*k*/) +{ + return 0; +} -} // namespace schedule +constexpr std::size_t +compute_loss_on(const tiled_scheduler_local & /*sched*/, std::size_t /*n_tiles*/, std::size_t /*k*/) +{ + return 0; +} GPRAT_NS_END diff --git a/core/include/gprat/scheduler/cyclic.hpp b/core/include/gprat/scheduler/cyclic.hpp index 71f58d77..9afcf934 100644 --- a/core/include/gprat/scheduler/cyclic.hpp +++ b/core/include/gprat/scheduler/cyclic.hpp @@ -43,115 +43,111 @@ struct tiled_scheduler_cyclic : tiled_scheduler_distributed std::size_t height; }; -namespace schedule -{ - constexpr std::size_t -covariance_tile(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +covariance_tile_on(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) { return (row % sched.height) + (col % sched.width); } constexpr std::size_t -cross_covariance_tile(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +cross_covariance_tile_on(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) { return (row % sched.height) + (col % sched.width); } -constexpr std::size_t alpha_tile(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t i) +constexpr std::size_t alpha_tile_on(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t i) { return (i % sched.height) + (i % sched.width); } -constexpr std::size_t prediction_tile(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t i) +constexpr std::size_t prediction_tile_on(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t i) { return (i % sched.height) + (i % sched.width); } -constexpr std::size_t cholesky_potrf(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k) +constexpr std::size_t cholesky_potrf_on(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k) { return (k % sched.height) + (k % sched.width); } -constexpr std::size_t cholesky_syrk(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t m) +constexpr std::size_t cholesky_syrk_on(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t m) { return (m % sched.height) + (m % sched.width); } constexpr std::size_t -cholesky_trsm(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) +cholesky_trsm_on(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) { return (m % sched.height) + (k % sched.width); } -constexpr std::size_t -cholesky_gemm(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m, std::size_t n) +constexpr std::size_t cholesky_gemm_on( + const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m, std::size_t n) { return (m % sched.height) + (n % sched.width); } -constexpr std::size_t solve_trsv(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +constexpr std::size_t solve_trsv_on(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) { return (k % sched.height) + (k % sched.width); } -constexpr std::size_t solve_trsm(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k) +constexpr std::size_t solve_trsm_on(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k) { return (k % sched.height) + (k % sched.width); } constexpr std::size_t -solve_gemv(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) +solve_gemv_on(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) { return (k % sched.height) + (m % sched.width); } constexpr std::size_t -solve_matrix_trsm(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t c, std::size_t k) +solve_matrix_trsm_on(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t c, std::size_t k) { return (k % sched.height) + (c % sched.width); } -constexpr std::size_t solve_matrix_gemm( +constexpr std::size_t solve_matrix_gemm_on( const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t c, std::size_t k, std::size_t m) { return (m % sched.height) + (c % sched.width); } constexpr std::size_t -multiply_gemv(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) +multiply_gemv_on(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) { return (k % sched.height) + (m % sched.width); } -constexpr std::size_t k_rank_dot_diag_syrk(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k) +constexpr std::size_t +k_rank_dot_diag_syrk_on(const tiled_scheduler_cyclic &sched, std::size_t /*n_tiles*/, std::size_t k) { return (k % sched.height) + (k % sched.width); } constexpr std::size_t -k_rank_gemm(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) +k_rank_gemm_on(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t c, std::size_t k, std::size_t m) { return (k * n_tiles + m) % sched.num_localities; } -constexpr std::size_t vector_axpy(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +constexpr std::size_t vector_axpy_on(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) { return (k * n_tiles + k) % sched.num_localities; } -constexpr std::size_t get_diagonal(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +constexpr std::size_t get_diagonal_on(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) { return (k * n_tiles + k) % sched.num_localities; } -constexpr std::size_t compute_loss(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) +constexpr std::size_t compute_loss_on(const tiled_scheduler_cyclic &sched, std::size_t n_tiles, std::size_t k) { return (k * n_tiles + k) % sched.num_localities; } -} // namespace schedule - GPRAT_NS_END #endif diff --git a/core/include/gprat/scheduler/sma.hpp b/core/include/gprat/scheduler/sma.hpp index e43e9d1d..5d5a4a5a 100644 --- a/core/include/gprat/scheduler/sma.hpp +++ b/core/include/gprat/scheduler/sma.hpp @@ -16,160 +16,155 @@ struct tiled_scheduler_sma : tiled_scheduler_distributed std::size_t num_localities = localities_.size(); }; -namespace schedule -{ - constexpr std::size_t -covariance_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +covariance_tile_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) { return (row + col) % sched.num_localities; } constexpr std::size_t -cross_covariance_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +cross_covariance_tile_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) { return (row + col) % sched.num_localities; } -constexpr std::size_t alpha_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) +constexpr std::size_t alpha_tile_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) { return (2 * i) % sched.num_localities; } -constexpr std::size_t prediction_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) +constexpr std::size_t prediction_tile_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) { return (2 * i) % sched.num_localities; } constexpr std::size_t -t_cross_covariance_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +t_cross_covariance_tile_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) { return (row + col) % sched.num_localities; } constexpr std::size_t -prior_K_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +prior_K_tile_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) { return (row + col) % sched.num_localities; } constexpr std::size_t -K_inv_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +K_inv_tile_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) { return (row + col) % sched.num_localities; } constexpr std::size_t -K_grad_v_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +K_grad_v_tile_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) { return (row + col) % sched.num_localities; } constexpr std::size_t -K_grad_l_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) +K_grad_l_tile_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t row, std::size_t col) { return (row + col) % sched.num_localities; } -constexpr std::size_t uncertainty_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) +constexpr std::size_t uncertainty_tile_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) { return (2 * i) % sched.num_localities; } -constexpr std::size_t inter_alpha_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) +constexpr std::size_t inter_alpha_tile_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) { return (2 * i) % sched.num_localities; } -constexpr std::size_t diag_tile(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) +constexpr std::size_t diag_tile_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t i) { return i % sched.num_localities; } -constexpr std::size_t cholesky_potrf(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) +constexpr std::size_t cholesky_potrf_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) { return (2 * k) % sched.num_localities; } -constexpr std::size_t cholesky_syrk(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t m) +constexpr std::size_t cholesky_syrk_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t m) { return (2 * m) % sched.num_localities; } constexpr std::size_t -cholesky_trsm(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) +cholesky_trsm_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) { return (k + m) % sched.num_localities; } constexpr std::size_t -cholesky_gemm(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m, std::size_t n) +cholesky_gemm_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m, std::size_t n) { return (m + n) % sched.num_localities; } -constexpr std::size_t solve_trsv(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) +constexpr std::size_t solve_trsv_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) { return (2 * k) % sched.num_localities; } -constexpr std::size_t solve_trsm(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) +constexpr std::size_t solve_trsm_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) { return (2 * k) % sched.num_localities; } constexpr std::size_t -solve_gemv(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) +solve_gemv_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) { return (k + m) % sched.num_localities; } constexpr std::size_t -solve_matrix_trsm(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t c, std::size_t k) +solve_matrix_trsm_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t c, std::size_t k) { return (k + c) % sched.num_localities; } -constexpr std::size_t solve_matrix_gemm( +constexpr std::size_t solve_matrix_gemm_on( const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t c, std::size_t /*k*/, std::size_t m) { return (c + m) % sched.num_localities; } constexpr std::size_t -multiply_gemv(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) +multiply_gemv_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k, std::size_t m) { return (k + m) % sched.num_localities; } -constexpr std::size_t k_rank_dot_diag_syrk(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) +constexpr std::size_t k_rank_dot_diag_syrk_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) { return (2 * k) % sched.num_localities; } -constexpr std::size_t -k_rank_gemm(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t /*c*/, std::size_t k, std::size_t m) +constexpr std::size_t k_rank_gemm_on( + const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t /*c*/, std::size_t k, std::size_t m) { return (k + m) % sched.num_localities; } -constexpr std::size_t vector_axpy(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) +constexpr std::size_t vector_axpy_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) { return (2 * k) % sched.num_localities; } -constexpr std::size_t get_diagonal(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) +constexpr std::size_t get_diagonal_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) { return (2 * k) % sched.num_localities; } -constexpr std::size_t compute_loss(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) +constexpr std::size_t compute_loss_on(const tiled_scheduler_sma &sched, std::size_t /*n_tiles*/, std::size_t k) { return (2 * k) % sched.num_localities; } -} // namespace schedule - GPRAT_NS_END #endif diff --git a/test/src/output_correctness.cpp b/test/src/output_correctness.cpp index 52aa5569..a2546106 100644 --- a/test/src/output_correctness.cpp +++ b/test/src/output_correctness.cpp @@ -205,7 +205,7 @@ TEST_CASE("GP GPU results match known-good values (no loss)", "[integration][gpu { if (!gprat::compiled_with_cuda()) { - WARN("CUDA not available — skipping GPU test."); + WARN("CUDA not available — skipping GPU test."); return; }