Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 4 additions & 8 deletions .github/workflows/test-and-lint.yml
Original file line number Diff line number Diff line change
Expand Up @@ -26,10 +26,8 @@ jobs:
- name: Install system dependencies
run: |
sudo apt-get -qq update
sudo apt-get install -y ca-certificates libopenblas-dev gcc-9 g++-9
sudo update-alternatives \
--install /usr/bin/gcc gcc /usr/bin/gcc-9 60 \
--slave /usr/bin/g++ g++ /usr/bin/g++-9
sudo apt-get install -y ca-certificates libopenblas-dev build-essential
g++ --version

- name: Build and install aihwkit wheel
run: |
Expand Down Expand Up @@ -60,10 +58,8 @@ jobs:
- name: Install system dependencies
run: |
sudo apt-get -qq update
sudo apt-get install -y ca-certificates libopenblas-dev gcc-9 g++-9
sudo update-alternatives \
--install /usr/bin/gcc gcc /usr/bin/gcc-9 60 \
--slave /usr/bin/g++ g++ /usr/bin/g++-9
sudo apt-get install -y ca-certificates libopenblas-dev build-essential
g++ --version

- name: Build and install aihwkit wheel
run: |
Expand Down
1 change: 1 addition & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,7 @@ The format is based on [Keep a Changelog], and this project adheres to
* Remove reset bias clamp in rpu_pulsed_device.h [(\#761)](https://github.com/IBM/aihwkit/pull/761)

### Fixed
* Build against `torch >= 2.14`, whose headers require C++20: the C++ standard is now detected from the installed torch headers (C++17 for `torch <= 2.13`, C++20 for `torch >= 2.14`) and can be overridden with the new `RPU_CXX_STANDARD` cmake option
* Improve Python executable detection and error handling in CMake [(\#757)](https://github.com/IBM/aihwkit/pull/757)
* Retune PWU kernel when m_batch grows after initial m=1 update [(\#763)](https://github.com/IBM/aihwkit/pull/763)
* Tests for dynamic tolerance for cuDNN TF32 precision on Ampere+ GPUs [(\#771)](https://github.com/IBM/aihwkit/pull/771)
Expand Down
27 changes: 16 additions & 11 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -21,16 +21,14 @@ option(RPU_USE_FASTMOD "Use fast mod" OFF)
option(RPU_USE_FASTRAND "Use fastrand" OFF)
option(RPU_USE_TORCH_BUFFERS "Use torch buffers for RPUCuda" ON)

# C++ standard used for every target (17 or 20). Leave empty to pick it up from
# the installed torch headers (see cmake/dependencies.cmake): torch <= 2.13
# builds with C++17, torch >= 2.14 requires C++20.
set(RPU_CXX_STANDARD "" CACHE STRING "C++ standard (17, 20). Empty: detect from torch headers")

set(RPU_BLAS "OpenBLAS" CACHE STRING "BLAS backend of choice (OpenBLAS, MKL)")
set(RPU_CUDA_ARCHITECTURES "75;80;89" CACHE STRING "Target CUDA architectures")

# Internal variables.
set(CUDA_TARGET_PROPERTIES POSITION_INDEPENDENT_CODE ON
CUDA_RESOLVE_DEVICE_SYMBOLS ON
CUDA_SEPARABLE_COMPILATION ON
CXX_STANDARD 17)

# Append the virtualenv library path to cmake.
if(DEFINED ENV{VIRTUAL_ENV})
include_directories("$ENV{VIRTUAL_ENV}/include")
Expand All @@ -43,10 +41,17 @@ include(cmake/dependencies.cmake)
include(cmake/dependencies_cuda.cmake)
include(cmake/dependencies_test.cmake)

# Internal variables.
set(CUDA_TARGET_PROPERTIES POSITION_INDEPENDENT_CODE ON
CUDA_RESOLVE_DEVICE_SYMBOLS ON
CUDA_SEPARABLE_COMPILATION ON
CXX_STANDARD ${RPU_CXX_STANDARD}
CUDA_STANDARD ${RPU_CXX_STANDARD})

# Set compilation flags.
if(WIN32)
set(CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} /O2")
set(CMAKE_CXX_STANDARD 17)
set(CMAKE_CXX_STANDARD ${RPU_CXX_STANDARD})
set(CMAKE_CXX_STANDARD_REQUIRED ON)
else()
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -Wall -Wno-narrowing -Wno-strict-overflow")
Expand All @@ -68,7 +73,7 @@ if(WIN32)
target_link_libraries(RPU_CPU c10.lib torch_cpu.lib)
endif()

set_target_properties(RPU_CPU PROPERTIES CXX_STANDARD 17
set_target_properties(RPU_CPU PROPERTIES CXX_STANDARD ${RPU_CXX_STANDARD}
POSITION_INDEPENDENT_CODE ON)

add_compile_definitions(RPU_USE_WITH_TORCH)
Expand Down Expand Up @@ -156,7 +161,7 @@ if (BUILD_EXTENSION)

add_library(AIHWKIT_EXTENSION_OPS ${AIHWKIT_EXTENSION_OPS_CPU_SRCS})

set_target_properties(AIHWKIT_EXTENSION_OPS PROPERTIES CXX_STANDARD 17
set_target_properties(AIHWKIT_EXTENSION_OPS PROPERTIES CXX_STANDARD ${RPU_CXX_STANDARD}
POSITION_INDEPENDENT_CODE ON)

target_link_libraries(AIHWKIT_EXTENSION_OPS
Expand Down Expand Up @@ -198,7 +203,7 @@ if (BUILD_EXTENSION)
target_link_libraries(${extension_module_name} PRIVATE torch_python)
target_include_directories(${extension_module_name} PRIVATE src/aihwkit/extension/extension_src)
target_include_directories(${extension_module_name} PRIVATE src/aihwkit/extension/extension_src/ops)
set_target_properties(${extension_module_name} PROPERTIES CXX_STANDARD 17
set_target_properties(${extension_module_name} PROPERTIES CXX_STANDARD ${RPU_CXX_STANDARD}
POSITION_INDEPENDENT_CODE ON)

if (USE_CUDA)
Expand Down Expand Up @@ -231,7 +236,7 @@ if(BUILD_TEST)
add_executable(${test_name} ${test_src})
target_link_libraries(${test_name} gtest gmock)
target_link_libraries(${test_name} torch_python c10 torch_cpu)
set_target_properties(${test_name} PROPERTIES CXX_STANDARD 17
set_target_properties(${test_name} PROPERTIES CXX_STANDARD ${RPU_CXX_STANDARD}
POSITION_INDEPENDENT_CODE ON)

if(WIN32)
Expand Down
33 changes: 33 additions & 0 deletions cmake/dependencies.cmake
Original file line number Diff line number Diff line change
Expand Up @@ -116,6 +116,39 @@ find_package(Torch REQUIRED)
include_directories(${TORCH_INCLUDE_DIRS})
link_directories(${TORCH_LIB_DIR})

# C++ standard. torch's headers carry a hard requirement on the language
# standard (ATen/ATen.h: `#if __cplusplus < 202002L` / `< 201703L` followed by
# an #error): releases up to 2.13 build with C++17, 2.14 and later need C++20.
# Read the number out of that guard so the standard follows whichever torch is
# installed, unless RPU_CXX_STANDARD was given explicitly.
if(NOT RPU_CXX_STANDARD)
set(RPU_CXX_STANDARD 17)
set(_rpu_aten_h "")
foreach(_dir ${TORCH_INCLUDE_DIRS})
if(EXISTS "${_dir}/ATen/ATen.h")
set(_rpu_aten_h "${_dir}/ATen/ATen.h")
break()
endif()
endforeach()
if(_rpu_aten_h)
file(STRINGS "${_rpu_aten_h}" _rpu_aten_guard
REGEX "__cplusplus[ \t]*<[ \t]*20[0-9][0-9][0-9][0-9]L" LIMIT_COUNT 1)
if(_rpu_aten_guard MATCHES "__cplusplus[ \t]*<[ \t]*20([0-9][0-9])[0-9][0-9]L")
set(RPU_CXX_STANDARD ${CMAKE_MATCH_1})
endif()
endif()
message(STATUS "Using C++${RPU_CXX_STANDARD} (required by ${_rpu_aten_h})")
else()
message(STATUS "Using C++${RPU_CXX_STANDARD} (from RPU_CXX_STANDARD)")
endif()

if(RPU_CXX_STANDARD GREATER 17 AND CMAKE_CXX_COMPILER_ID STREQUAL "GNU"
AND CMAKE_CXX_COMPILER_VERSION VERSION_LESS 10)
message(WARNING "C++${RPU_CXX_STANDARD} is required by the installed torch, but GCC "
"${CMAKE_CXX_COMPILER_VERSION} has no usable C++20 support (concepts, <compare>). "
"Use GCC 10 or newer, or install torch <= 2.13.")
endif()

if (CMAKE_COMPILER_IS_GNUCXX)
# Prefer ABI from Torch CMake config (works even when Python build isolation
# makes `import torch` unavailable in helper subprocesses).
Expand Down
12 changes: 9 additions & 3 deletions docs/source/developer_install.rst
Original file line number Diff line number Diff line change
Expand Up @@ -218,16 +218,22 @@ Compilation flags
There are several ``cmake`` options that can be used for customizing the
compilation process:

========================== ================================================ =======
========================== ================================================ ==============
Flag Description Default
========================== ================================================ =======
========================== ================================================ ==============
``USE_CUDA`` Build with CUDA support ``OFF``
``BUILD_TEST`` Build the C++ test binaries ``OFF``
``RPU_BLAS`` BLAS backend of choice (``OpenBLAS`` or ``MKL``) ``OpenBLAS``
``RPU_USE_FASTMOD`` Use fast mod ``ON``
``RPU_USE_FASTRAND`` Use fastrand ``OFF``
``RPU_CUDA_ARCHITECTURES`` Target CUDA architectures ``60;70;75;80``
========================== ================================================ =======
``RPU_CXX_STANDARD`` C++ standard (``17`` or ``20``) auto-detected
========================== ================================================ ==============

The C++ standard is taken from the installed ``PyTorch`` headers when
``RPU_CXX_STANDARD`` is not set: ``torch`` releases up to ``2.13`` are built
with C++17, ``torch >= 2.14`` requires C++20 (and hence a C++20-capable
compiler, e.g. ``gcc >= 10`` or ``CUDA >= 12`` for the GPU build).

The options can be passed both to ``setuptools`` or to ``cmake`` directly. For
example, for compiling and installing with CUDA support::
Expand Down
4 changes: 2 additions & 2 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -18,8 +18,8 @@ dependencies = [
"scikit-build >= 0.18.1",
"scikit-learn",
"pybind11 >= 3.0.1",
"torch == 2.13.0",
"torchvision == 0.28.0",
"torch",
"torchvision",
"scipy",
"requests >= 2.32, <3",
"numpy",
Expand Down
2 changes: 1 addition & 1 deletion src/aihwkit/simulator/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@ set(python_module_name rpu_base)
file(GLOB RPU_BINDINGS_SRCS rpu_base_src/*.cpp)
pybind11_add_module(${python_module_name} MODULE ${RPU_BINDINGS_SRCS})
target_link_libraries(${python_module_name} PRIVATE torch_python)
set_target_properties(${python_module_name} PROPERTIES CXX_STANDARD 17)
set_target_properties(${python_module_name} PROPERTIES CXX_STANDARD ${RPU_CXX_STANDARD})

if (USE_CUDA)
target_link_libraries(${python_module_name} PRIVATE RPU_GPU)
Expand Down
2 changes: 1 addition & 1 deletion src/rpucuda/rpu_linearstep_device.h
Original file line number Diff line number Diff line change
Expand Up @@ -69,7 +69,7 @@ BUILD_PULSED_DEVICE_META_PARAMETER(

template <typename T>
struct SoftBoundsRPUDeviceMetaParameter : LinearStepRPUDeviceMetaParameter<T> {
SoftBoundsRPUDeviceMetaParameter<T>() {
SoftBoundsRPUDeviceMetaParameter() {
// ctor, to overwrite default values to softbounds
this->ls_decrease_up = (T)1.0; // meaning 0 update at bounds
this->ls_decrease_down = (T)1.0; // meaning 0 update at bounds
Expand Down
Loading