diff --git a/.github/workflows/test-and-lint.yml b/.github/workflows/test-and-lint.yml index 1d7a1fe61..d5fdf970e 100644 --- a/.github/workflows/test-and-lint.yml +++ b/.github/workflows/test-and-lint.yml @@ -26,10 +26,8 @@ jobs: - name: Install system dependencies run: | sudo apt-get -qq update - sudo apt-get install -y ca-certificates libopenblas-dev gcc-9 g++-9 - sudo update-alternatives \ - --install /usr/bin/gcc gcc /usr/bin/gcc-9 60 \ - --slave /usr/bin/g++ g++ /usr/bin/g++-9 + sudo apt-get install -y ca-certificates libopenblas-dev build-essential + g++ --version - name: Build and install aihwkit wheel run: | @@ -60,10 +58,8 @@ jobs: - name: Install system dependencies run: | sudo apt-get -qq update - sudo apt-get install -y ca-certificates libopenblas-dev gcc-9 g++-9 - sudo update-alternatives \ - --install /usr/bin/gcc gcc /usr/bin/gcc-9 60 \ - --slave /usr/bin/g++ g++ /usr/bin/g++-9 + sudo apt-get install -y ca-certificates libopenblas-dev build-essential + g++ --version - name: Build and install aihwkit wheel run: | diff --git a/CHANGELOG.md b/CHANGELOG.md index 11e9b42bb..7e47874ad 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,6 +25,7 @@ The format is based on [Keep a Changelog], and this project adheres to * Remove reset bias clamp in rpu_pulsed_device.h [(\#761)](https://github.com/IBM/aihwkit/pull/761) ### Fixed +* Build against `torch >= 2.14`, whose headers require C++20: the C++ standard is now detected from the installed torch headers (C++17 for `torch <= 2.13`, C++20 for `torch >= 2.14`) and can be overridden with the new `RPU_CXX_STANDARD` cmake option * Improve Python executable detection and error handling in CMake [(\#757)](https://github.com/IBM/aihwkit/pull/757) * Retune PWU kernel when m_batch grows after initial m=1 update [(\#763)](https://github.com/IBM/aihwkit/pull/763) * Tests for dynamic tolerance for cuDNN TF32 precision on Ampere+ GPUs [(\#771)](https://github.com/IBM/aihwkit/pull/771) diff --git a/CMakeLists.txt b/CMakeLists.txt index efd5a3588..11ff94f0b 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -21,16 +21,14 @@ option(RPU_USE_FASTMOD "Use fast mod" OFF) option(RPU_USE_FASTRAND "Use fastrand" OFF) option(RPU_USE_TORCH_BUFFERS "Use torch buffers for RPUCuda" ON) +# C++ standard used for every target (17 or 20). Leave empty to pick it up from +# the installed torch headers (see cmake/dependencies.cmake): torch <= 2.13 +# builds with C++17, torch >= 2.14 requires C++20. +set(RPU_CXX_STANDARD "" CACHE STRING "C++ standard (17, 20). Empty: detect from torch headers") set(RPU_BLAS "OpenBLAS" CACHE STRING "BLAS backend of choice (OpenBLAS, MKL)") set(RPU_CUDA_ARCHITECTURES "75;80;89" CACHE STRING "Target CUDA architectures") -# Internal variables. -set(CUDA_TARGET_PROPERTIES POSITION_INDEPENDENT_CODE ON - CUDA_RESOLVE_DEVICE_SYMBOLS ON - CUDA_SEPARABLE_COMPILATION ON - CXX_STANDARD 17) - # Append the virtualenv library path to cmake. if(DEFINED ENV{VIRTUAL_ENV}) include_directories("$ENV{VIRTUAL_ENV}/include") @@ -43,10 +41,17 @@ include(cmake/dependencies.cmake) include(cmake/dependencies_cuda.cmake) include(cmake/dependencies_test.cmake) +# Internal variables. +set(CUDA_TARGET_PROPERTIES POSITION_INDEPENDENT_CODE ON + CUDA_RESOLVE_DEVICE_SYMBOLS ON + CUDA_SEPARABLE_COMPILATION ON + CXX_STANDARD ${RPU_CXX_STANDARD} + CUDA_STANDARD ${RPU_CXX_STANDARD}) + # Set compilation flags. if(WIN32) set(CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} /O2") - set(CMAKE_CXX_STANDARD 17) + set(CMAKE_CXX_STANDARD ${RPU_CXX_STANDARD}) set(CMAKE_CXX_STANDARD_REQUIRED ON) else() set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -Wall -Wno-narrowing -Wno-strict-overflow") @@ -68,7 +73,7 @@ if(WIN32) target_link_libraries(RPU_CPU c10.lib torch_cpu.lib) endif() -set_target_properties(RPU_CPU PROPERTIES CXX_STANDARD 17 +set_target_properties(RPU_CPU PROPERTIES CXX_STANDARD ${RPU_CXX_STANDARD} POSITION_INDEPENDENT_CODE ON) add_compile_definitions(RPU_USE_WITH_TORCH) @@ -156,7 +161,7 @@ if (BUILD_EXTENSION) add_library(AIHWKIT_EXTENSION_OPS ${AIHWKIT_EXTENSION_OPS_CPU_SRCS}) - set_target_properties(AIHWKIT_EXTENSION_OPS PROPERTIES CXX_STANDARD 17 + set_target_properties(AIHWKIT_EXTENSION_OPS PROPERTIES CXX_STANDARD ${RPU_CXX_STANDARD} POSITION_INDEPENDENT_CODE ON) target_link_libraries(AIHWKIT_EXTENSION_OPS @@ -198,7 +203,7 @@ if (BUILD_EXTENSION) target_link_libraries(${extension_module_name} PRIVATE torch_python) target_include_directories(${extension_module_name} PRIVATE src/aihwkit/extension/extension_src) target_include_directories(${extension_module_name} PRIVATE src/aihwkit/extension/extension_src/ops) - set_target_properties(${extension_module_name} PROPERTIES CXX_STANDARD 17 + set_target_properties(${extension_module_name} PROPERTIES CXX_STANDARD ${RPU_CXX_STANDARD} POSITION_INDEPENDENT_CODE ON) if (USE_CUDA) @@ -231,7 +236,7 @@ if(BUILD_TEST) add_executable(${test_name} ${test_src}) target_link_libraries(${test_name} gtest gmock) target_link_libraries(${test_name} torch_python c10 torch_cpu) - set_target_properties(${test_name} PROPERTIES CXX_STANDARD 17 + set_target_properties(${test_name} PROPERTIES CXX_STANDARD ${RPU_CXX_STANDARD} POSITION_INDEPENDENT_CODE ON) if(WIN32) diff --git a/cmake/dependencies.cmake b/cmake/dependencies.cmake index 62e592867..528676d0b 100644 --- a/cmake/dependencies.cmake +++ b/cmake/dependencies.cmake @@ -116,6 +116,39 @@ find_package(Torch REQUIRED) include_directories(${TORCH_INCLUDE_DIRS}) link_directories(${TORCH_LIB_DIR}) +# C++ standard. torch's headers carry a hard requirement on the language +# standard (ATen/ATen.h: `#if __cplusplus < 202002L` / `< 201703L` followed by +# an #error): releases up to 2.13 build with C++17, 2.14 and later need C++20. +# Read the number out of that guard so the standard follows whichever torch is +# installed, unless RPU_CXX_STANDARD was given explicitly. +if(NOT RPU_CXX_STANDARD) + set(RPU_CXX_STANDARD 17) + set(_rpu_aten_h "") + foreach(_dir ${TORCH_INCLUDE_DIRS}) + if(EXISTS "${_dir}/ATen/ATen.h") + set(_rpu_aten_h "${_dir}/ATen/ATen.h") + break() + endif() + endforeach() + if(_rpu_aten_h) + file(STRINGS "${_rpu_aten_h}" _rpu_aten_guard + REGEX "__cplusplus[ \t]*<[ \t]*20[0-9][0-9][0-9][0-9]L" LIMIT_COUNT 1) + if(_rpu_aten_guard MATCHES "__cplusplus[ \t]*<[ \t]*20([0-9][0-9])[0-9][0-9]L") + set(RPU_CXX_STANDARD ${CMAKE_MATCH_1}) + endif() + endif() + message(STATUS "Using C++${RPU_CXX_STANDARD} (required by ${_rpu_aten_h})") +else() + message(STATUS "Using C++${RPU_CXX_STANDARD} (from RPU_CXX_STANDARD)") +endif() + +if(RPU_CXX_STANDARD GREATER 17 AND CMAKE_CXX_COMPILER_ID STREQUAL "GNU" + AND CMAKE_CXX_COMPILER_VERSION VERSION_LESS 10) + message(WARNING "C++${RPU_CXX_STANDARD} is required by the installed torch, but GCC " + "${CMAKE_CXX_COMPILER_VERSION} has no usable C++20 support (concepts, ). " + "Use GCC 10 or newer, or install torch <= 2.13.") +endif() + if (CMAKE_COMPILER_IS_GNUCXX) # Prefer ABI from Torch CMake config (works even when Python build isolation # makes `import torch` unavailable in helper subprocesses). diff --git a/docs/source/developer_install.rst b/docs/source/developer_install.rst index 489108561..af6d46fd7 100644 --- a/docs/source/developer_install.rst +++ b/docs/source/developer_install.rst @@ -218,16 +218,22 @@ Compilation flags There are several ``cmake`` options that can be used for customizing the compilation process: -========================== ================================================ ======= +========================== ================================================ ============== Flag Description Default -========================== ================================================ ======= +========================== ================================================ ============== ``USE_CUDA`` Build with CUDA support ``OFF`` ``BUILD_TEST`` Build the C++ test binaries ``OFF`` ``RPU_BLAS`` BLAS backend of choice (``OpenBLAS`` or ``MKL``) ``OpenBLAS`` ``RPU_USE_FASTMOD`` Use fast mod ``ON`` ``RPU_USE_FASTRAND`` Use fastrand ``OFF`` ``RPU_CUDA_ARCHITECTURES`` Target CUDA architectures ``60;70;75;80`` -========================== ================================================ ======= +``RPU_CXX_STANDARD`` C++ standard (``17`` or ``20``) auto-detected +========================== ================================================ ============== + +The C++ standard is taken from the installed ``PyTorch`` headers when +``RPU_CXX_STANDARD`` is not set: ``torch`` releases up to ``2.13`` are built +with C++17, ``torch >= 2.14`` requires C++20 (and hence a C++20-capable +compiler, e.g. ``gcc >= 10`` or ``CUDA >= 12`` for the GPU build). The options can be passed both to ``setuptools`` or to ``cmake`` directly. For example, for compiling and installing with CUDA support:: diff --git a/pyproject.toml b/pyproject.toml index c6bcd5003..c2be7f6d5 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -18,8 +18,8 @@ dependencies = [ "scikit-build >= 0.18.1", "scikit-learn", "pybind11 >= 3.0.1", - "torch == 2.13.0", - "torchvision == 0.28.0", + "torch", + "torchvision", "scipy", "requests >= 2.32, <3", "numpy", diff --git a/src/aihwkit/simulator/CMakeLists.txt b/src/aihwkit/simulator/CMakeLists.txt index bf4963887..bd9395f13 100644 --- a/src/aihwkit/simulator/CMakeLists.txt +++ b/src/aihwkit/simulator/CMakeLists.txt @@ -7,7 +7,7 @@ set(python_module_name rpu_base) file(GLOB RPU_BINDINGS_SRCS rpu_base_src/*.cpp) pybind11_add_module(${python_module_name} MODULE ${RPU_BINDINGS_SRCS}) target_link_libraries(${python_module_name} PRIVATE torch_python) -set_target_properties(${python_module_name} PROPERTIES CXX_STANDARD 17) +set_target_properties(${python_module_name} PROPERTIES CXX_STANDARD ${RPU_CXX_STANDARD}) if (USE_CUDA) target_link_libraries(${python_module_name} PRIVATE RPU_GPU) diff --git a/src/rpucuda/rpu_linearstep_device.h b/src/rpucuda/rpu_linearstep_device.h index 0b5e735d2..08d90555d 100644 --- a/src/rpucuda/rpu_linearstep_device.h +++ b/src/rpucuda/rpu_linearstep_device.h @@ -69,7 +69,7 @@ BUILD_PULSED_DEVICE_META_PARAMETER( template struct SoftBoundsRPUDeviceMetaParameter : LinearStepRPUDeviceMetaParameter { - SoftBoundsRPUDeviceMetaParameter() { + SoftBoundsRPUDeviceMetaParameter() { // ctor, to overwrite default values to softbounds this->ls_decrease_up = (T)1.0; // meaning 0 update at bounds this->ls_decrease_down = (T)1.0; // meaning 0 update at bounds