Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions CMakeModules/AFcuda_helpers.cmake
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,8 @@
# The complete license agreement can be obtained at:
# http://arrayfire.com/licenses/BSD-3-Clause

include(select_compute_arch)

find_program(NVPRUNE NAMES nvprune)
cuda_select_nvcc_arch_flags(cuda_architecture_flags ${CUDA_architecture_build_targets})
set(cuda_architecture_flags ${cuda_architecture_flags} CACHE INTERNAL "CUDA compute flags" FORCE)
Expand Down
4 changes: 2 additions & 2 deletions CMakeModules/select_compute_arch.cmake
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@
# ARCH_AND_PTX : NAME | NUM.NUM | NUM.NUM(NUM.NUM) | NUM.NUM+PTX
# NAME: Fermi Kepler Maxwell Kepler+Tegra Kepler+Tesla Maxwell+Tegra Pascal Volta Turing Ampere
# NUM: Any number. Only those pairs are currently accepted by NVCC though:
# 2.0 2.1 3.0 3.2 3.5 3.7 5.0 5.2 5.3 6.0 6.2 7.0 7.2 7.5 8.0 8.6 9.0
# 2.0 2.1 3.0 3.2 3.5 3.7 5.0 5.2 5.3 6.0 6.2 7.0 7.2 7.5 8.0 8.6 8.9 9.0 12.0
# Returns LIST of flags to be added to CUDA_NVCC_FLAGS in ${out_variable}
# Additionally, sets ${out_variable}_readable to the resulting numeric list
# Example:
Expand Down Expand Up @@ -229,7 +229,7 @@ function(CUDA_SELECT_NVCC_ARCH_FLAGS out_variable)
set(add_ptx TRUE)
set(arch_name ${CMAKE_MATCH_1})
endif()
if(arch_name MATCHES "^([0-9]\\.[0-9](\\([0-9]\\.[0-9]\\))?)$")
if(arch_name MATCHES "^([0-9]+\\.[0-9]+(\\([0-9]+\\.[0-9]+\\))?)$")
set(arch_bin ${CMAKE_MATCH_1})
set(arch_ptx ${arch_bin})
else()
Expand Down
12 changes: 9 additions & 3 deletions src/backend/cuda/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -858,7 +858,9 @@ if(AF_INSTALL_STANDALONE)
endif()

if(WIN32 OR NOT AF_WITH_STATIC_CUDA_NUMERIC_LIBS)
if(CUDA_VERSION_MAJOR VERSION_EQUAL 12)
if(CUDA_VERSION_MAJOR VERSION_EQUAL 13)
afcu_collect_libs(cufft LIB_MAJOR 12 LIB_MINOR 1)
elseif(CUDA_VERSION_MAJOR VERSION_EQUAL 12)
afcu_collect_libs(cufft LIB_MAJOR 11 LIB_MINOR 3)
elseif(CUDA_VERSION_MAJOR VERSION_EQUAL 11)
afcu_collect_libs(cufft LIB_MAJOR 10 LIB_MINOR 4)
Expand All @@ -869,7 +871,9 @@ if(AF_INSTALL_STANDALONE)
if(CUDA_VERSION VERSION_GREATER 10.0)
afcu_collect_libs(cublasLt)
endif()
if(CUDA_VERSION_MAJOR VERSION_EQUAL 12)
if(CUDA_VERSION_MAJOR VERSION_EQUAL 13)
afcu_collect_libs(cusolver LIB_MAJOR 12 LIB_MINOR 0)
elseif(CUDA_VERSION_MAJOR VERSION_EQUAL 12)
afcu_collect_libs(cusolver LIB_MAJOR 11 LIB_MINOR 7)
else()
afcu_collect_libs(cusolver)
Expand All @@ -879,7 +883,9 @@ if(AF_INSTALL_STANDALONE)
afcu_collect_libs(nvJitLink)
endif()
elseif(NOT ${use_static_cuda_lapack})
if(CUDA_VERSION_MAJOR VERSION_EQUAL 12)
if(CUDA_VERSION_MAJOR VERSION_EQUAL 13)
afcu_collect_libs(cusolver LIB_MAJOR 12 LIB_MINOR 0)
elseif(CUDA_VERSION_MAJOR VERSION_EQUAL 12)
afcu_collect_libs(cusolver LIB_MAJOR 11 LIB_MINOR 7)
else()
afcu_collect_libs(cusolver)
Expand Down
6 changes: 6 additions & 0 deletions src/backend/cuda/cufft.cu
Original file line number Diff line number Diff line change
Expand Up @@ -38,19 +38,25 @@ const char *_cufftGetResultString(cufftResult res) {

case CUFFT_UNALIGNED_DATA: return "cuFFT: unaligned data (deprecated)";

#ifdef CUFFT_INCOMPLETE_PARAMETER_LIST
case CUFFT_INCOMPLETE_PARAMETER_LIST:
return "cuFFT: call is missing parameters";
#endif

case CUFFT_INVALID_DEVICE:
return "cuFFT: plan execution different than plan creation";

#ifdef CUFFT_PARSE_ERROR
case CUFFT_PARSE_ERROR: return "cuFFT: plan parse error";
#endif

case CUFFT_NO_WORKSPACE: return "cuFFT: no workspace provided";

case CUFFT_NOT_IMPLEMENTED: return "cuFFT: not implemented";

#ifdef CUFFT_LICENSE_ERROR
case CUFFT_LICENSE_ERROR: return "cuFFT: license error";
#endif

#if CUDA_VERSION >= 8000
case CUFFT_NOT_SUPPORTED: return "cuFFT: not supported";
Expand Down
6 changes: 5 additions & 1 deletion src/backend/cuda/device_manager.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -598,9 +598,13 @@ DeviceManager::DeviceManager()
AF_TRACE("Unsuppored device: {}", dev.prop.name);
continue;
} else {
int clock_rate_khz = 0;
CUDA_CHECK(
cudaDeviceGetAttribute(&clock_rate_khz, cudaDevAttrClockRate,
i));
dev.flops = static_cast<size_t>(dev.prop.multiProcessorCount) *
compute2cores(dev.prop.major, dev.prop.minor) *
dev.prop.clockRate;
static_cast<size_t>(clock_rate_khz);
dev.nativeId = i;
AF_TRACE(
"Found device: {} (sm_{}{}) ({:0.3} GB | ~{} GFLOPs | {} "
Expand Down
2 changes: 1 addition & 1 deletion src/backend/cuda/kernel/regions.hpp
Original file line number Diff line number Diff line change
Expand Up @@ -341,7 +341,7 @@ __global__ static void update_equiv(arrayfire::cuda::Param<T> equiv_map,
}

template<typename T>
struct clamp_to_one : public thrust::unary_function<T, T> {
struct clamp_to_one {
__host__ __device__ T operator()(const T& in) const {
return (in >= (T)1) ? (T)1 : in;
}
Expand Down
Loading