From b51a758d8054e8563700d8015146f7d7910d3dd3 Mon Sep 17 00:00:00 2001 From: Matthew Johnson Date: Fri, 24 Apr 2015 00:58:14 +0100 Subject: [PATCH 01/14] Skeleton for the CUDA native provider DLL project. --- MathNet.Numerics.NativeProviders.sln | 20 ++++ src/NativeProviders/CUDA/blas.c | 0 src/NativeProviders/CUDA/capabilities.cpp | 0 src/NativeProviders/CUDA/lapack.cpp | 0 src/NativeProviders/CUDA/memory.c | 0 src/NativeProviders/CUDA/vector_functions.c | 0 .../Windows/CUDA/CUDAWrapper.vcxproj | 96 +++++++++++++++++++ .../Windows/CUDA/CUDAWrapper.vcxproj.filters | 42 ++++++++ 8 files changed, 158 insertions(+) create mode 100644 src/NativeProviders/CUDA/blas.c create mode 100644 src/NativeProviders/CUDA/capabilities.cpp create mode 100644 src/NativeProviders/CUDA/lapack.cpp create mode 100644 src/NativeProviders/CUDA/memory.c create mode 100644 src/NativeProviders/CUDA/vector_functions.c create mode 100644 src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj create mode 100644 src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters diff --git a/MathNet.Numerics.NativeProviders.sln b/MathNet.Numerics.NativeProviders.sln index 141b7074..232e00b6 100644 --- a/MathNet.Numerics.NativeProviders.sln +++ b/MathNet.Numerics.NativeProviders.sln @@ -20,6 +20,8 @@ Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "Numerics", "src\Numerics\Nu EndProject Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "UnitTests-MKL", "src\UnitTests\UnitTests-MKL.csproj", "{3515A344-AB5F-41C7-A14C-04A79B3FFAB1}" EndProject +Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "CUDA", "src\NativeProviders\Windows\CUDA\CUDAWrapper.vcxproj", "{5A52B796-7F41-4C90-8DE2-F3F391C4482C}" +EndProject Global GlobalSection(SolutionConfigurationPlatforms) = preSolution Debug|Any CPU = Debug|Any CPU @@ -111,6 +113,24 @@ Global {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-Signed|Mixed Platforms.Build.0 = Release|Any CPU {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-Signed|Win32.ActiveCfg = Release|Any CPU {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-Signed|x64.ActiveCfg = Release|Any CPU + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Debug|Any CPU.ActiveCfg = Debug|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Debug|Mixed Platforms.ActiveCfg = Debug|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Debug|Mixed Platforms.Build.0 = Debug|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Debug|Win32.ActiveCfg = Debug|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Debug|Win32.Build.0 = Debug|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Debug|x64.ActiveCfg = Debug|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|Any CPU.ActiveCfg = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|Mixed Platforms.ActiveCfg = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|Mixed Platforms.Build.0 = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|Win32.ActiveCfg = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|Win32.Build.0 = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|x64.ActiveCfg = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|Any CPU.ActiveCfg = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|Mixed Platforms.ActiveCfg = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|Mixed Platforms.Build.0 = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|Win32.ActiveCfg = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|Win32.Build.0 = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|x64.ActiveCfg = Release|Win32 EndGlobalSection GlobalSection(SolutionProperties) = preSolution HideSolutionNode = FALSE diff --git a/src/NativeProviders/CUDA/blas.c b/src/NativeProviders/CUDA/blas.c new file mode 100644 index 00000000..e69de29b diff --git a/src/NativeProviders/CUDA/capabilities.cpp b/src/NativeProviders/CUDA/capabilities.cpp new file mode 100644 index 00000000..e69de29b diff --git a/src/NativeProviders/CUDA/lapack.cpp b/src/NativeProviders/CUDA/lapack.cpp new file mode 100644 index 00000000..e69de29b diff --git a/src/NativeProviders/CUDA/memory.c b/src/NativeProviders/CUDA/memory.c new file mode 100644 index 00000000..e69de29b diff --git a/src/NativeProviders/CUDA/vector_functions.c b/src/NativeProviders/CUDA/vector_functions.c new file mode 100644 index 00000000..e69de29b diff --git a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj new file mode 100644 index 00000000..22dc1a85 --- /dev/null +++ b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj @@ -0,0 +1,96 @@ + + + + + Debug + Win32 + + + Release + Win32 + + + + + + + + + + + + + + + {5A52B796-7F41-4C90-8DE2-F3F391C4482C} + CUDA + CUDA + + + + DynamicLibrary + true + v120 + MultiByte + + + DynamicLibrary + false + v120 + true + MultiByte + + + + + + + + + + + + + $(ProjectDir)..\..\..\..\out\CUDA\Windows\x86\ + $(Platform)\$(Configuration)\ + MathNet.Numerics.CUDA + + + $(ProjectDir)..\..\..\..\out\CUDA\Windows\x86\ + $(Platform)\$(Configuration)\ + MathNet.Numerics.CUDA + + + + Level3 + Disabled + true + $(CUDA_PATH)\include;$(ProjectDir)..\..\Common;$(ProjectDir)..\..\CUDA;%(AdditionalIncludeDirectories) + + + true + cublas.lib;%(AdditionalDependencies) + $(CUDA_PATH)\lib\x64;%(AdditionalLibraryDirectories) + + + + + Level3 + MaxSpeed + true + true + true + $(CUDA_PATH)\include;$(ProjectDir)..\..\Common;$(ProjectDir)..\..\CUDA;%(AdditionalIncludeDirectories) + + + true + true + true + cublas.lib;%(AdditionalDependencies) + $(CUDA_PATH)\lib\x64;%(AdditionalLibraryDirectories) + + + + + + \ No newline at end of file diff --git a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters new file mode 100644 index 00000000..f834db07 --- /dev/null +++ b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters @@ -0,0 +1,42 @@ + + + + + {4FC737F1-C7A5-4376-A066-2A32D752A2FF} + cpp;c;cc;cxx;def;odl;idl;hpj;bat;asm;asmx + + + {93995380-89BD-4b04-88EB-625FBE52EBFB} + h;hh;hpp;hxx;hm;inl;inc;xsd + + + {67DA6AB6-F800-4c08-8B7A-83BB121AAD01} + rc;ico;cur;bmp;dlg;rc2;rct;bin;rgs;gif;jpg;jpeg;jpe;resx;tiff;tif;png;wav;mfcribbon-ms + + + + + Resource Files + + + + + Source Files + + + Source Files + + + Source Files + + + Source Files + + + Source Files + + + Source Files + + + \ No newline at end of file From bf1f3b81523673d1a3b4ce4b987710d32fbf57e6 Mon Sep 17 00:00:00 2001 From: Matthew Johnson Date: Fri, 24 Apr 2015 02:18:22 +0100 Subject: [PATCH 02/14] Adding in some naive first stabs at integration. --- MathNet.Numerics.NativeProviders.sln | 3 +- src/NativeProviders/CUDA/blas.c | 93 ++++ src/NativeProviders/CUDA/lapack.cpp | 521 ++++++++++++++++++ src/NativeProviders/CUDA/vector_functions.c | 0 .../Windows/CUDA/CUDAWrapper.vcxproj | 70 ++- .../Windows/CUDA/CUDAWrapper.vcxproj.filters | 3 - 6 files changed, 684 insertions(+), 6 deletions(-) delete mode 100644 src/NativeProviders/CUDA/vector_functions.c diff --git a/MathNet.Numerics.NativeProviders.sln b/MathNet.Numerics.NativeProviders.sln index 232e00b6..1c7bf18d 100644 --- a/MathNet.Numerics.NativeProviders.sln +++ b/MathNet.Numerics.NativeProviders.sln @@ -118,7 +118,8 @@ Global {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Debug|Mixed Platforms.Build.0 = Debug|Win32 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Debug|Win32.ActiveCfg = Debug|Win32 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Debug|Win32.Build.0 = Debug|Win32 - {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Debug|x64.ActiveCfg = Debug|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Debug|x64.ActiveCfg = Debug|x64 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Debug|x64.Build.0 = Debug|x64 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|Any CPU.ActiveCfg = Release|Win32 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|Mixed Platforms.ActiveCfg = Release|Win32 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|Mixed Platforms.Build.0 = Release|Win32 diff --git a/src/NativeProviders/CUDA/blas.c b/src/NativeProviders/CUDA/blas.c index e69de29b..18ff81aa 100644 --- a/src/NativeProviders/CUDA/blas.c +++ b/src/NativeProviders/CUDA/blas.c @@ -0,0 +1,93 @@ +#include "cublas_v2.h" +#include "wrapper_common.h" + +#if GCC +extern "C" { +#endif + DLLEXPORT void s_axpy(const cublasHandle_t handle, const int n, const float alpha, const float x[], float y[]){ + cublasSaxpy(handle, n, &alpha, x, 1, y, 1); + } + + DLLEXPORT void d_axpy(const cublasHandle_t handle, const int n, const double alpha, const double x[], double y[]){ + cublasDaxpy(handle, n, &alpha, x, 1, y, 1); + } + + DLLEXPORT void c_axpy(const cublasHandle_t handle, const int n, const cuComplex alpha, const cuComplex x[], cuComplex y[]){ + cublasCaxpy(handle, n, &alpha, x, 1, y, 1); + } + + DLLEXPORT void z_axpy(const cublasHandle_t handle, const int n, const cuDoubleComplex alpha, const cuDoubleComplex x[], cuDoubleComplex y[]){ + cublasZaxpy(handle, n, &alpha, x, 1, y, 1); + } + + DLLEXPORT void s_scale(const cublasHandle_t handle, const int n, const float alpha, float x[]){ + cublasSscal(handle, n, &alpha, x, 1); + } + + DLLEXPORT void d_scale(const cublasHandle_t handle, const int n, const double alpha, double x[]){ + cublasDscal(handle, n, &alpha, x, 1); + } + + DLLEXPORT void c_scale(const cublasHandle_t handle, const int n, const cuComplex alpha, cuComplex x[]){ + cublasCscal(handle, n, &alpha, x, 1); + } + + DLLEXPORT void z_scale(const cublasHandle_t handle, const int n, const cuDoubleComplex alpha, cuDoubleComplex x[]){ + cublasZscal(handle, n, &alpha, x, 1); + } + + DLLEXPORT float s_dot_product(const cublasHandle_t handle, const int n, const float x[], const float y[]){ + float ret; + cublasSdot(handle, n, x, 1, y, 1, &ret); + return ret; + } + + DLLEXPORT double d_dot_product(const cublasHandle_t handle, const int n, const double x[], const double y[]){ + double ret; + cublasDdot(handle, n, x, 1, y, 1, &ret); + return ret; + } + + DLLEXPORT cuComplex c_dot_product(const cublasHandle_t handle, const int n, const cuComplex x[], const cuComplex y[]){ + cuComplex ret; + cublasCdotu(handle, n, x, 1, y, 1, &ret); + return ret; + } + + DLLEXPORT cuDoubleComplex z_dot_product(const cublasHandle_t handle, const int n, const cuDoubleComplex x[], const cuDoubleComplex y[]){ + cuDoubleComplex ret; + cublasZdotu(handle, n, x, 1, y, 1, &ret); + return ret; + } + + DLLEXPORT void s_matrix_multiply(const cublasHandle_t handle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const float alpha, const float x[], const float y[], const float beta, float c[]){ + int lda = transA == CUBLAS_OP_N ? m : k; + int ldb = transB == CUBLAS_OP_N ? k : n; + + cublasSgemm(handle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m); + } + + DLLEXPORT void d_matrix_multiply(const cublasHandle_t handle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const double alpha, const double x[], const double y[], const double beta, double c[]){ + int lda = transA == CUBLAS_OP_N ? m : k; + int ldb = transB == CUBLAS_OP_N ? k : n; + + cublasDgemm(handle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m); + } + + DLLEXPORT void c_matrix_multiply(const cublasHandle_t handle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const cuComplex alpha, const cuComplex x[], const cuComplex y[], const cuComplex beta, cuComplex c[]){ + int lda = transA == CUBLAS_OP_N ? m : k; + int ldb = transB == CUBLAS_OP_N ? k : n; + + cublasCgemm(handle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m); + } + + DLLEXPORT void z_matrix_multiply(const cublasHandle_t handle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const cuDoubleComplex alpha, const cuDoubleComplex x[], const cuDoubleComplex y[], const cuDoubleComplex beta, cuDoubleComplex c[]){ + int lda = transA == CUBLAS_OP_N ? m : k; + int ldb = transB == CUBLAS_OP_N ? k : n; + + cublasZgemm(handle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m); + } + +#if GCC +} +#endif diff --git a/src/NativeProviders/CUDA/lapack.cpp b/src/NativeProviders/CUDA/lapack.cpp index e69de29b..189a0b4c 100644 --- a/src/NativeProviders/CUDA/lapack.cpp +++ b/src/NativeProviders/CUDA/lapack.cpp @@ -0,0 +1,521 @@ +#include "lapack_common.h" +#include "wrapper_common.h" +#include "cublas.h" +#include "cusolverDn.h" +#include + +template +inline int lu_factor(int m, T a[], int ipiv[], + int(*getrf) (CBLAS_ORDER, const int, const int, K*, const int, int*)) +{ + int info = getrf(CblasColMajor, m, m, a, m, ipiv); + shift_ipiv_down(m, ipiv); + return info; +}; + +template +inline int lu_inverse(int n, T a[], + int(*getrf) (CBLAS_ORDER, const int, const int, K*, const int, int*), + int(*getri) (CBLAS_ORDER, const int, K*, const int, const int*)) +{ + int* ipiv = new int[n]; + int info = getrf(CblasColMajor, n, n, a, n, ipiv); + + if (info != 0){ + delete[] ipiv; + return info; + } + + info = getri(CblasColMajor, n, a, n, ipiv); + delete[] ipiv; + return info; +}; + +template +inline int lu_inverse_factored(int n, T a[], int ipiv[], + int(*getri) (CBLAS_ORDER, const int, K*, const int, const int*)) +{ + shift_ipiv_up(n, ipiv); + int info = getri(CblasColMajor, n, a, n, ipiv); + shift_ipiv_down(n, ipiv); + return info; +} + +template +inline int lu_solve_factored(int n, int nrhs, T a[], int ipiv[], T b[], + int(*getrs) (CBLAS_ORDER, CBLAS_TRANSPOSE, const int, const int, const K*, const int, const int*, K*, const int)) +{ + shift_ipiv_up(n, ipiv); + int info = getrs(CblasColMajor, CblasNoTrans, n, nrhs, a, n, ipiv, b, n); + shift_ipiv_down(n, ipiv); + return info; +} + +template +inline int lu_solve(int n, int nrhs, T a[], T b[], + int(*getrf) (CBLAS_ORDER, const int, const int, K*, const int, int*), + int(*getrs) (CBLAS_ORDER, CBLAS_TRANSPOSE, const int, const int, const K*, const int, const int*, K*, const int)) +{ + T* clone = Clone(n, n, a); + int* ipiv = new int[n]; + int info = getrf(CblasColMajor, n, n, clone, n, ipiv); + + if (info != 0){ + delete[] ipiv; + delete[] clone; + return info; + } + + info = getrs(CblasColMajor, CblasNoTrans, n, nrhs, clone, n, ipiv, b, n); + delete[] ipiv; + delete[] clone; + return info; +} + +template +inline int cholesky_factor(int n, T* a, int(*potrf) (CBLAS_ORDER, CBLAS_UPLO, const int, K*, const int)) +{ + int info = potrf(CblasColMajor, CblasLower, n, a, n); + T zero = T(); + for (int i = 0; i < n; ++i) + { + int index = i * n; + for (int j = 0; j < n && i > j; ++j) + { + a[index + j] = zero; + } + } + return info; +} + +template +inline int cholesky_solve(int n, int nrhs, T a[], T b[], + int(*potrf) (CBLAS_ORDER, CBLAS_UPLO, const int, K*, const int), + int(*potrs) (CBLAS_ORDER, CBLAS_UPLO, const int, const int, const K*, const int, K*, const int)) +{ + T* clone = Clone(n, n, a); + int info = potrf(CblasColMajor, CblasLower, n, clone, n); + + if (info != 0){ + delete[] clone; + return info; + } + + info = potrs(CblasColMajor, CblasLower, n, nrhs, clone, n, b, n); + delete[] clone; + return info; +} + +template +inline int cholesky_solve_factored(int n, int nrhs, T a[], T b[], + int(*potrs) (CBLAS_ORDER, CBLAS_UPLO, const int, const int, const K*, const int, K*, const int)) +{ + return potrs(CblasColMajor, CblasLower, n, nrhs, a, n, b, n); +} + +template +inline int qr_factor(int m, int n, T r[], T tau[], T q[], T work[], int len, + int(*geqrf) (const int, const int, K*, const int, T*), + int(*orgqr) (const int, const int, const int, K*, const int, const K*)) +{ + int info = geqrf(m, n, r, m, tau); + + for (int i = 0; i < m; ++i) + { + for (int j = 0; j < m && j < n; ++j) + { + if (i > j) + { + q[j * m + i] = r[j * m + i]; + } + } + } + + //compute the q elements explicitly + if (m <= n) + { + info = orgqr(m, m, m, q, m, tau); + } + else + { + info = orgqr(m, m, n, q, m, tau); + } + + return info; +} + +template +inline int qr_thin_factor(int m, int n, T q[], T tau[], T r[], T work[], int len, + void(*geqrf) (const int*, const int*, T*, const int*, T*, T*, const int*, int*), + void(*orgqr) (const int*, const int*, const int*, T*, const int*, const T*, T*, const int*, int*)) +{ + int info = 0; + geqrf(&m, &n, q, &m, tau, work, &len, &info); + + for (int i = 0; i < n; ++i) + { + for (int j = 0; j < n; ++j) + { + if (i <= j) { + r[j * n + i] = q[j * m + i]; + } + } + } + + orgqr(&m, &n, &n, q, &m, tau, work, &len, &info); + + return info; +} + +template +inline int qr_solve(int m, int n, int bn, T a[], T b[], T x[], T work[], int len, + void(*gels) (const char*, const int*, const int*, const int*, T*, + const int*, T* b, const int*, T*, const int*, int*)) +{ + T* clone_a = new T[m*n]; + std::memcpy(clone_a, a, m*n*sizeof(T)); + + T* clone_b = new T[m*bn]; + std::memcpy(clone_b, b, m*bn*sizeof(T)); + + char N = 'N'; + int info = 0; + gels(&N, &m, &n, &bn, clone_a, &m, clone_b, &m, work, &len, &info); + copyBtoX(n, n, bn, clone_b, x); + + delete[] clone_a; + delete[] clone_b; + return info; +} + +template +inline int qr_solve_factored(int m, int n, int bn, T r[], T b[], T tau[], T x[], T work[], int len, + void(*ormqr) (const char*, const char*, const int*, const int*, const int*, + const T*, const int*, const T*, T*, const int*, T*, const int*, int* info), + void(*trsm) (const CBLAS_ORDER, const CBLAS_SIDE, const CBLAS_UPLO, const CBLAS_TRANSPOSE, const CBLAS_DIAG, + const int, const int, const T, const T*, const int, T*, const int)) +{ + T* clone_b = new T[m*bn]; + std::memcpy(clone_b, b, m*bn*sizeof(T)); + + char side = 'L'; + char tran = 'T'; + int info = 0; + ormqr(&side, &tran, &m, &bn, &n, r, &m, tau, clone_b, &m, work, &len, &info); + trsm(CblasColMajor, CblasLeft, CblasUpper, CblasNoTrans, CblasNonUnit, n, bn, 1.0, r, m, clone_b, m); + copyBtoX(n, n, bn, clone_b, x); + + delete[] clone_b; + return info; +} + +template +inline int complex_qr_solve_factored(int m, int n, int bn, T r[], T b[], T tau[], T x[], T work[], int len, + void(*unmqr) (const char*, const char*, const int*, const int*, const int*, + const T*, const int*, const T*, T*, const int*, T*, const int*, int* info), + void(*trsm) (const CBLAS_ORDER, const CBLAS_SIDE, const CBLAS_UPLO, const CBLAS_TRANSPOSE, const CBLAS_DIAG, + const int, const int, const void*, const void*, const int, void*, const int ldb)) +{ + T* clone_b = new T[m*bn]; + std::memcpy(clone_b, b, m*bn*sizeof(T)); + + char side = 'L'; + char tran = 'C'; + int info = 0; + unmqr(&side, &tran, &m, &bn, &n, r, &m, tau, clone_b, &m, work, &len, &info); + + T one = { 1.0f, 0.0f }; + trsm(CblasColMajor, CblasLeft, CblasUpper, CblasNoTrans, CblasNonUnit, n, bn, &one, r, m, clone_b, m); + copyBtoX(n, n, bn, clone_b, x); + + delete[] clone_b; + return info; +} + +template +inline int svd_factor(bool compute_vectors, int m, int n, T a[], T s[], T u[], T v[], T work[], int len, + void(*gesvd) (const char*, const char*, const int*, const int*, T*, const int*, + T*, T*, const int*, T*, const int*, T*, const int*, int*)) +{ + int info = 0; + char job = compute_vectors ? 'A' : 'N'; + gesvd(&job, &job, &m, &n, a, &m, s, u, &m, v, &n, work, &len, &info); + return info; +} + + +template +inline int complex_svd_factor(bool compute_vectors, int m, int n, T a[], T s[], T u[], T v[], T work[], int len, + void(*gesvd) (const char*, const char*, const int*, const int*, T*, const int*, + R*, T*, const int*, T*, const int*, T*, const int*, R*, int*)) +{ + int info = 0; + int dim_s = std::min(m, n); + R* rwork = new R[5 * dim_s]; + R* s_local = new R[dim_s]; + char job = compute_vectors ? 'A' : 'N'; + gesvd(&job, &job, &m, &n, a, &m, s_local, u, &m, v, &n, work, &len, rwork, &info); + + for (int index = 0; index < dim_s; ++index){ + T value = { s_local[index], 0.0f }; + s[index] = value; + } + + delete[] rwork; + delete[] s_local; + return info; +} + +extern "C" { + DLLEXPORT int s_lu_factor(int m, float a[], int ipiv[]) { + return lu_factor(m, a, ipiv, cusolverDnSgetrf); + } + + DLLEXPORT int d_lu_factor(int m, double a[], int ipiv[]) { + return lu_factor(m, a, ipiv, cusolverDnDgetrf); + } + + DLLEXPORT int c_lu_factor(int m, cuComplex a[], int ipiv[]) { + return lu_factor(m, a, ipiv, cusolverDnCgetrf); + } + + DLLEXPORT int z_lu_factor(int m, cuDoubleComplex a[], int ipiv[]) { + return lu_factor(m, a, ipiv, cusolverDnZgetrf); + } + + DLLEXPORT int s_lu_inverse(int n, float a[]) + { + return lu_inverse(n, a, cusolverDnSgetrf, cusolverDnSgetri); + } + + DLLEXPORT int d_lu_inverse(int n, double a[]) + { + return lu_inverse(n, a, cusolverDnDgetrf, cusolverDnDgetri); + } + + DLLEXPORT int c_lu_inverse(int n, cuComplex a[]) + { + return lu_inverse(n, a, cusolverDnCgetrf, cusolverDnCgetri); + } + + DLLEXPORT int z_lu_inverse(int n, cuDoubleComplex a[]) + { + return lu_inverse(n, a, cusolverDnZgetrf, cusolverDnZgetri); + } + + DLLEXPORT int s_lu_inverse_factored(int n, float a[], int ipiv[], float work[], int lwork) + { + return lu_inverse_factored(n, a, ipiv, cusolverDnSgetri); + } + + DLLEXPORT int d_lu_inverse_factored(int n, double a[], int ipiv[], double work[], int lwork) + { + return lu_inverse_factored(n, a, ipiv, cusolverDnDgetri); + } + + DLLEXPORT int c_lu_inverse_factored(int n, cuComplex a[], int ipiv[], cuComplex work[], int lwork) + { + return lu_inverse_factored(n, a, ipiv, cusolverDnCgetri); + } + + DLLEXPORT int z_lu_inverse_factored(int n, cuDoubleComplex a[], int ipiv[], cuDoubleComplex work[], int lwork) + { + return lu_inverse_factored(n, a, ipiv, cusolverDnZgetri); + } + + DLLEXPORT int s_lu_solve_factored(int n, int nrhs, float a[], int ipiv[], float b[]) + { + return lu_solve_factored(n, nrhs, a, ipiv, b, cusolverDnSgetrs); + } + + DLLEXPORT int d_lu_solve_factored(int n, int nrhs, double a[], int ipiv[], double b[]) + { + return lu_solve_factored(n, nrhs, a, ipiv, b, cusolverDnDgetrs); + } + + DLLEXPORT int c_lu_solve_factored(int n, int nrhs, cuComplex a[], int ipiv[], cuComplex b[]) + { + return lu_solve_factored(n, nrhs, a, ipiv, b, cusolverDnCgetrs); + } + + DLLEXPORT int z_lu_solve_factored(int n, int nrhs, cuDoubleComplex a[], int ipiv[], cuDoubleComplex b[]) + { + return lu_solve_factored(n, nrhs, a, ipiv, b, cusolverDnZgetrs); + } + + DLLEXPORT int s_lu_solve(int n, int nrhs, float a[], float b[]) + { + return lu_solve(n, nrhs, a, b, cusolverDnSgetrf, cusolverDnSgetrs); + } + + DLLEXPORT int d_lu_solve(int n, int nrhs, double a[], double b[]) + { + return lu_solve(n, nrhs, a, b, cusolverDnDgetrf, cusolverDnDgetrs); + } + + DLLEXPORT int c_lu_solve(int n, int nrhs, cuComplex a[], cuComplex b[]) + { + return lu_solve(n, nrhs, a, b, cusolverDnCgetrf, cusolverDnCgetrs); + } + + DLLEXPORT int z_lu_solve(int n, int nrhs, cuDoubleComplex a[], cuDoubleComplex b[]) + { + return lu_solve(n, nrhs, a, b, cusolverDnZgetrf, cusolverDnZgetrs); + } + + DLLEXPORT int s_cholesky_factor(int n, float a[]){ + return cholesky_factor(n, a, cusolverDnSpotrf); + } + + DLLEXPORT int d_cholesky_factor(int n, double* a){ + return cholesky_factor(n, a, cusolverDnDpotrf); + } + + DLLEXPORT int c_cholesky_factor(int n, cuComplex a[]){ + return cholesky_factor(n, a, cusolverDnCpotrf); + } + + DLLEXPORT int z_cholesky_factor(int n, cuDoubleComplex a[]){ + return cholesky_factor(n, a, cusolverDnZpotrf); + } + + DLLEXPORT int s_cholesky_solve(int n, int nrhs, float a[], float b[]) + { + return cholesky_solve(n, nrhs, a, b, cusolverDnSpotrf, cusolverDnSpotrs); + } + + DLLEXPORT int d_cholesky_solve(int n, int nrhs, double a[], double b[]) + { + return cholesky_solve(n, nrhs, a, b, cusolverDnDpotrf, cusolverDnDpotrs); + } + + DLLEXPORT int c_cholesky_solve(int n, int nrhs, cuComplex a[], cuComplex b[]) + { + return cholesky_solve(n, nrhs, a, b, cusolverDnCpotrf, cusolverDnCpotrs); + } + + DLLEXPORT int z_cholesky_solve(int n, int nrhs, cuDoubleComplex a[], cuDoubleComplex b[]) + { + return cholesky_solve(n, nrhs, a, b, cusolverDnZpotrf, cusolverDnZpotrs); + } + + DLLEXPORT int s_cholesky_solve_factored(int n, int nrhs, float a[], float b[]) + { + return cholesky_solve_factored(n, nrhs, a, b, cusolverDnSpotrs); + } + + DLLEXPORT int d_cholesky_solve_factored(int n, int nrhs, double a[], double b[]) + { + return cholesky_solve_factored(n, nrhs, a, b, cusolverDnDpotrs); + } + + DLLEXPORT int c_cholesky_solve_factored(int n, int nrhs, cuComplex a[], cuComplex b[]) + { + return cholesky_solve_factored(n, nrhs, a, b, cusolverDnCpotrs); + } + + DLLEXPORT int z_cholesky_solve_factored(int n, int nrhs, cuDoubleComplex a[], cuDoubleComplex b[]) + { + return cholesky_solve_factored(n, nrhs, a, b, cusolverDnZpotrs); + } + + /*DLLEXPORT int s_qr_factor(int m, int n, float r[], float tau[], float q[], float work[], int len) + { + return qr_factor(m, n, r, tau, q, work, len, cusolverDnSgeqrf, cusolverDnSorgqr); + } + + DLLEXPORT int s_qr_thin_factor(int m, int n, float q[], float tau[], float r[], float work[], int len) + { + return qr_thin_factor(m, n, q, tau, r, work, len, cusolverDnSgeqrf, cusolverDnSorgqr); + } + + DLLEXPORT int d_qr_factor(int m, int n, double r[], double tau[], double q[], double work[], int len) + { + return qr_factor(m, n, r, tau, q, work, len, cusolverDnDgeqrf, cusolverDnDorgqr); + } + + DLLEXPORT int d_qr_thin_factor(int m, int n, double q[], double tau[], double r[], double work[], int len) + { + return qr_thin_factor(m, n, q, tau, r, work, len, cusolverDnDgeqrf, cusolverDnDorgqr); + } + + DLLEXPORT int c_qr_factor(int m, int n, cuComplex r[], cuComplex tau[], cuComplex q[], cuComplex work[], int len) + { + return qr_factor(m, n, r, tau, q, work, len, cusolverDnCgeqrf, cusolverDnCungqr); + } + + DLLEXPORT int c_qr_thin_factor(int m, int n, cuComplex q[], cuComplex tau[], cuComplex r[], cuComplex work[], int len) + { + return qr_thin_factor(m, n, q, tau, r, work, len, cusolverDnCgeqrf, cusolverDnCungqr); + } + + DLLEXPORT int z_qr_factor(int m, int n, cuDoubleComplex r[], cuDoubleComplex tau[], cuDoubleComplex q[]) + { + return qr_factor(m, n, r, tau, q, work, len, cusolverDnZgeqrf, cusolverDnZungqr); + } + + DLLEXPORT int z_qr_thin_factor(int m, int n, cuDoubleComplex q[], cuDoubleComplex tau[], cuDoubleComplex r[]) + { + return qr_thin_factor(m, n, q, tau, r, work, len, cusolverDnZgeqrf, cusolverDnZungqr); + } + + DLLEXPORT int s_qr_solve(int m, int n, int bn, float a[], float b[], float x[], float work[], int len) + { + return qr_solve(m, n, bn, a, b, x, work, len, sgels); + } + + DLLEXPORT int d_qr_solve(int m, int n, int bn, double a[], double b[], double x[], double work[], int len) + { + return qr_solve(m, n, bn, a, b, x, work, len, dgels); + } + + DLLEXPORT int c_qr_solve(int m, int n, int bn, cuComplex a[], cuComplex b[], cuComplex x[], cuComplex work[], int len) + { + return qr_solve(m, n, bn, a, b, x, work, len, cgels); + } + + DLLEXPORT int z_qr_solve(int m, int n, int bn, cuDoubleComplex a[], cuDoubleComplex b[], cuDoubleComplex x[], cuDoubleComplex work[], int len) + { + return qr_solve(m, n, bn, a, b, x, work, len, zgels); + } + + DLLEXPORT int s_qr_solve_factored(int m, int n, int bn, float r[], float b[], float tau[], float x[], float work[], int len) + { + return qr_solve_factored(m, n, bn, r, b, tau, x, work, len, sormqr, cblas_strsm); + } + + DLLEXPORT int d_qr_solve_factored(int m, int n, int bn, double r[], double b[], double tau[], double x[], double work[], int len) + { + return qr_solve_factored(m, n, bn, r, b, tau, x, work, len, dormqr, cblas_dtrsm); + } + + DLLEXPORT int c_qr_solve_factored(int m, int n, int bn, cuComplex r[], cuComplex b[], cuComplex tau[], cuComplex x[], cuComplex work[], int len) + { + return complex_qr_solve_factored(m, n, bn, r, b, tau, x, work, len, cunmqr, cblas_ctrsm); + } + + DLLEXPORT int z_qr_solve_factored(int m, int n, int bn, cuDoubleComplex r[], cuDoubleComplex b[], cuDoubleComplex tau[], cuDoubleComplex x[], cuDoubleComplex work[], int len) + { + return complex_qr_solve_factored(m, n, bn, r, b, tau, x, work, len, zunmqr, cblas_ztrsm); + } + + DLLEXPORT int s_svd_factor(bool compute_vectors, int m, int n, float a[], float s[], float u[], float v[], float work[], int len) + { + return svd_factor(compute_vectors, m, n, a, s, u, v, work, len, sgesvd); + } + + DLLEXPORT int d_svd_factor(bool compute_vectors, int m, int n, double a[], double s[], double u[], double v[], double work[], int len) + { + return svd_factor(compute_vectors, m, n, a, s, u, v, work, len, dgesvd); + } + + DLLEXPORT int c_svd_factor(bool compute_vectors, int m, int n, cuComplex a[], cuComplex s[], cuComplex u[], cuComplex v[], cuComplex work[], int len) + { + return complex_svd_factor(compute_vectors, m, n, a, s, u, v, work, len, cgesvd); + } + + DLLEXPORT int z_svd_factor(bool compute_vectors, int m, int n, cuDoubleComplex a[], cuDoubleComplex s[], cuDoubleComplex u[], cuDoubleComplex v[], cuDoubleComplex work[], int len) + { + return complex_svd_factor(compute_vectors, m, n, a, s, u, v, work, len, zgesvd); + }*/ +} \ No newline at end of file diff --git a/src/NativeProviders/CUDA/vector_functions.c b/src/NativeProviders/CUDA/vector_functions.c deleted file mode 100644 index e69de29b..00000000 diff --git a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj index 22dc1a85..ec8944fd 100644 --- a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj +++ b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj @@ -5,10 +5,18 @@ Debug Win32 + + Debug + x64 + Release Win32 + + Release + x64 + @@ -19,7 +27,6 @@ - {5A52B796-7F41-4C90-8DE2-F3F391C4482C} @@ -33,6 +40,12 @@ v120 MultiByte + + DynamicLibrary + true + v120 + MultiByte + DynamicLibrary false @@ -40,26 +53,49 @@ true MultiByte + + DynamicLibrary + false + v120 + true + MultiByte + + + + + + + $(ProjectDir)..\..\..\..\out\CUDA\Windows\x86\ $(Platform)\$(Configuration)\ MathNet.Numerics.CUDA + + $(Platform)\$(Configuration)\ + MathNet.Numerics.CUDA + $(ProjectDir)..\..\..\..\out\CUDA\Windows\x64\ + $(ProjectDir)..\..\..\..\out\CUDA\Windows\x86\ $(Platform)\$(Configuration)\ MathNet.Numerics.CUDA + + $(Platform)\$(Configuration)\ + MathNet.Numerics.CUDA + $(ProjectDir)..\..\..\..\out\CUDA\Windows\x64\ + Level3 @@ -69,7 +105,20 @@ true - cublas.lib;%(AdditionalDependencies) + cublas.lib;cublas_device.lib;%(AdditionalDependencies) + $(CUDA_PATH)\lib\x64;%(AdditionalLibraryDirectories) + + + + + Level3 + Disabled + true + $(CUDA_PATH)\include;$(ProjectDir)..\..\Common;$(ProjectDir)..\..\CUDA;%(AdditionalIncludeDirectories) + + + true + cusolver.lib;cublas.lib;cublas_device.lib;%(AdditionalDependencies) $(CUDA_PATH)\lib\x64;%(AdditionalLibraryDirectories) @@ -90,6 +139,23 @@ $(CUDA_PATH)\lib\x64;%(AdditionalLibraryDirectories) + + + Level3 + MaxSpeed + true + true + true + $(CUDA_PATH)\include;$(ProjectDir)..\..\Common;$(ProjectDir)..\..\CUDA;%(AdditionalIncludeDirectories) + + + true + true + true + cusolver.lib;cublas.lib;%(AdditionalDependencies) + $(CUDA_PATH)\lib\x64;%(AdditionalLibraryDirectories) + + diff --git a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters index f834db07..0e998fbe 100644 --- a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters +++ b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters @@ -35,8 +35,5 @@ Source Files - - Source Files - \ No newline at end of file From 523309b1609243dca7432e20560b6a9231309b61 Mon Sep 17 00:00:00 2001 From: Matthew Johnson Date: Fri, 24 Apr 2015 20:19:01 +0100 Subject: [PATCH 03/14] Updating everything so that it has a decent chance at working. Moving on to the managed wrapper next. --- src/NativeProviders/CUDA/blas.c | 93 -- src/NativeProviders/CUDA/blas.cpp | 167 +++ src/NativeProviders/CUDA/lapack.cpp | 989 +++++++++++++----- .../Windows/CUDA/CUDAWrapper.vcxproj | 8 +- .../Windows/CUDA/CUDAWrapper.vcxproj.filters | 8 +- 5 files changed, 893 insertions(+), 372 deletions(-) delete mode 100644 src/NativeProviders/CUDA/blas.c create mode 100644 src/NativeProviders/CUDA/blas.cpp diff --git a/src/NativeProviders/CUDA/blas.c b/src/NativeProviders/CUDA/blas.c deleted file mode 100644 index 18ff81aa..00000000 --- a/src/NativeProviders/CUDA/blas.c +++ /dev/null @@ -1,93 +0,0 @@ -#include "cublas_v2.h" -#include "wrapper_common.h" - -#if GCC -extern "C" { -#endif - DLLEXPORT void s_axpy(const cublasHandle_t handle, const int n, const float alpha, const float x[], float y[]){ - cublasSaxpy(handle, n, &alpha, x, 1, y, 1); - } - - DLLEXPORT void d_axpy(const cublasHandle_t handle, const int n, const double alpha, const double x[], double y[]){ - cublasDaxpy(handle, n, &alpha, x, 1, y, 1); - } - - DLLEXPORT void c_axpy(const cublasHandle_t handle, const int n, const cuComplex alpha, const cuComplex x[], cuComplex y[]){ - cublasCaxpy(handle, n, &alpha, x, 1, y, 1); - } - - DLLEXPORT void z_axpy(const cublasHandle_t handle, const int n, const cuDoubleComplex alpha, const cuDoubleComplex x[], cuDoubleComplex y[]){ - cublasZaxpy(handle, n, &alpha, x, 1, y, 1); - } - - DLLEXPORT void s_scale(const cublasHandle_t handle, const int n, const float alpha, float x[]){ - cublasSscal(handle, n, &alpha, x, 1); - } - - DLLEXPORT void d_scale(const cublasHandle_t handle, const int n, const double alpha, double x[]){ - cublasDscal(handle, n, &alpha, x, 1); - } - - DLLEXPORT void c_scale(const cublasHandle_t handle, const int n, const cuComplex alpha, cuComplex x[]){ - cublasCscal(handle, n, &alpha, x, 1); - } - - DLLEXPORT void z_scale(const cublasHandle_t handle, const int n, const cuDoubleComplex alpha, cuDoubleComplex x[]){ - cublasZscal(handle, n, &alpha, x, 1); - } - - DLLEXPORT float s_dot_product(const cublasHandle_t handle, const int n, const float x[], const float y[]){ - float ret; - cublasSdot(handle, n, x, 1, y, 1, &ret); - return ret; - } - - DLLEXPORT double d_dot_product(const cublasHandle_t handle, const int n, const double x[], const double y[]){ - double ret; - cublasDdot(handle, n, x, 1, y, 1, &ret); - return ret; - } - - DLLEXPORT cuComplex c_dot_product(const cublasHandle_t handle, const int n, const cuComplex x[], const cuComplex y[]){ - cuComplex ret; - cublasCdotu(handle, n, x, 1, y, 1, &ret); - return ret; - } - - DLLEXPORT cuDoubleComplex z_dot_product(const cublasHandle_t handle, const int n, const cuDoubleComplex x[], const cuDoubleComplex y[]){ - cuDoubleComplex ret; - cublasZdotu(handle, n, x, 1, y, 1, &ret); - return ret; - } - - DLLEXPORT void s_matrix_multiply(const cublasHandle_t handle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const float alpha, const float x[], const float y[], const float beta, float c[]){ - int lda = transA == CUBLAS_OP_N ? m : k; - int ldb = transB == CUBLAS_OP_N ? k : n; - - cublasSgemm(handle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m); - } - - DLLEXPORT void d_matrix_multiply(const cublasHandle_t handle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const double alpha, const double x[], const double y[], const double beta, double c[]){ - int lda = transA == CUBLAS_OP_N ? m : k; - int ldb = transB == CUBLAS_OP_N ? k : n; - - cublasDgemm(handle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m); - } - - DLLEXPORT void c_matrix_multiply(const cublasHandle_t handle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const cuComplex alpha, const cuComplex x[], const cuComplex y[], const cuComplex beta, cuComplex c[]){ - int lda = transA == CUBLAS_OP_N ? m : k; - int ldb = transB == CUBLAS_OP_N ? k : n; - - cublasCgemm(handle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m); - } - - DLLEXPORT void z_matrix_multiply(const cublasHandle_t handle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const cuDoubleComplex alpha, const cuDoubleComplex x[], const cuDoubleComplex y[], const cuDoubleComplex beta, cuDoubleComplex c[]){ - int lda = transA == CUBLAS_OP_N ? m : k; - int ldb = transB == CUBLAS_OP_N ? k : n; - - cublasZgemm(handle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m); - } - -#if GCC -} -#endif diff --git a/src/NativeProviders/CUDA/blas.cpp b/src/NativeProviders/CUDA/blas.cpp new file mode 100644 index 00000000..47233b72 --- /dev/null +++ b/src/NativeProviders/CUDA/blas.cpp @@ -0,0 +1,167 @@ +#include "cublas_v2.h" +#include "cuda_runtime.h" +#include "wrapper_common.h" + +template +void cuda_axpy(const cublasHandle_t blasHandle, const int n, const T *alpha, const T x[], int incX, T y[], int incY, AXPY axpy) +{ + T *d_X = NULL; + T *d_Y = NULL; + cudaMalloc((void**)&d_X, n*sizeof(T)); + cudaMalloc((void**)&d_Y, n*sizeof(T)); + + cublasSetVector(n, sizeof(T), x, incX, d_X, incX); + cublasSetVector(n, sizeof(T), y, incY, d_Y, incY); + + axpy(blasHandle, n, alpha, d_X, incX, d_Y, incX); + + cublasGetVector(n, sizeof(T), d_Y, incY, y, incY); + + cudaFree(d_X); + cudaFree(d_Y); +} + +template +void cuda_scal(const cublasHandle_t blasHandle, const int n, const T *alpha, T x[], int incX, SCAL scal) +{ + T *d_X = NULL; + cudaMalloc((void**)&d_X, n*sizeof(T)); + + cublasSetVector(n, sizeof(T), x, incX, d_X, incX); + + scal(blasHandle, n, alpha, d_X, incX); + + cublasGetVector(n, sizeof(T), d_X, incX, x, incX); + + cudaFree(d_X); +} + +template +void cuda_dot(const cublasHandle_t blasHandle, const int n, const T x[], int incX, const T y[], int incY, T* result, DOT dot) +{ + T *d_X = NULL; + T *d_Y = NULL; + cudaMalloc((void**)&d_X, n*sizeof(T)); + cudaMalloc((void**)&d_Y, n*sizeof(T)); + + cublasSetVector(n, sizeof(T), x, incX, d_X, incX); + cublasSetVector(n, sizeof(T), y, incY, d_Y, incY); + + dot(blasHandle, n, d_X, incX, d_Y, incY, result); + + cudaFree(d_X); + cudaFree(d_Y); +} + +template +void cuda_gemm(const cublasHandle_t handle, const cublasOperation_t transa, const cublasOperation_t transb, int m, int n, int k, const T *alpha, const T A[], int lda, const T B[], int ldb, const T *beta, T C[], int ldc, GEMM gemm) +{ + T *d_A = NULL; + T *d_B = NULL; + T *d_C = NULL; + cudaMalloc((void**)&d_A, m*k*sizeof(T)); + cudaMalloc((void**)&d_B, k*n*sizeof(T)); + cudaMalloc((void**)&d_C, m*n*sizeof(T)); + + cublasSetMatrix(m, k, sizeof(T), A, m, d_A, m); + cublasSetMatrix(k, n, sizeof(T), B, k, d_B, k); + + gemm(handle, transa, transb, m, n, k, alpha, d_A, lda, d_B, ldb, beta, d_C, ldc); + + cublasGetMatrix(m, n, sizeof(T), d_C, m, C, m); + + cudaFree(d_A); + cudaFree(d_B); + cudaFree(d_C); +} + +#if GCC +extern "C" { +#endif + DLLEXPORT void s_axpy(const cublasHandle_t blasHandle, const int n, const float alpha, const float x[], float y[]){ + cuda_axpy(blasHandle, n, &alpha, x, 1, y, 1, cublasSaxpy); + } + + DLLEXPORT void d_axpy(const cublasHandle_t blasHandle, const int n, const double alpha, const double x[], double y[]){ + cuda_axpy(blasHandle, n, &alpha, x, 1, y, 1, cublasDaxpy); + } + + DLLEXPORT void c_axpy(const cublasHandle_t blasHandle, const int n, const cuComplex alpha, const cuComplex x[], cuComplex y[]){ + cuda_axpy(blasHandle, n, &alpha, x, 1, y, 1, cublasCaxpy); + } + + DLLEXPORT void z_axpy(const cublasHandle_t blasHandle, const int n, const cuDoubleComplex alpha, const cuDoubleComplex x[], cuDoubleComplex y[]){ + cuda_axpy(blasHandle, n, &alpha, x, 1, y, 1, cublasZaxpy); + } + + DLLEXPORT void s_scale(const cublasHandle_t blasHandle, const int n, const float alpha, float x[]){ + cuda_scal(blasHandle, n, &alpha, x, 1, cublasSscal); + } + + DLLEXPORT void d_scale(const cublasHandle_t blasHandle, const int n, const double alpha, double x[]){ + cuda_scal(blasHandle, n, &alpha, x, 1, cublasDscal); + } + + DLLEXPORT void c_scale(const cublasHandle_t blasHandle, const int n, const cuComplex alpha, cuComplex x[]){ + cuda_scal(blasHandle, n, &alpha, x, 1, cublasCscal); + } + + DLLEXPORT void z_scale(const cublasHandle_t blasHandle, const int n, const cuDoubleComplex alpha, cuDoubleComplex x[]){ + cuda_scal(blasHandle, n, &alpha, x, 1, cublasZscal); + } + + DLLEXPORT float s_dot_product(const cublasHandle_t blasHandle, const int n, const float x[], const float y[]){ + float ret; + cuda_dot(blasHandle, n, x, 1, y, 1, &ret, cublasSdot); + return ret; + } + + DLLEXPORT double d_dot_product(const cublasHandle_t blasHandle, const int n, const double x[], const double y[]){ + double ret; + cuda_dot(blasHandle, n, x, 1, y, 1, &ret, cublasDdot); + return ret; + } + + DLLEXPORT cuComplex c_dot_product(const cublasHandle_t blasHandle, const int n, const cuComplex x[], const cuComplex y[]){ + cuComplex ret; + cuda_dot(blasHandle, n, x, 1, y, 1, &ret, cublasCdotu); + return ret; + } + + DLLEXPORT cuDoubleComplex z_dot_product(const cublasHandle_t blasHandle, const int n, const cuDoubleComplex x[], const cuDoubleComplex y[]){ + cuDoubleComplex ret; + cuda_dot(blasHandle, n, x, 1, y, 1, &ret, cublasZdotu); + return ret; + } + + DLLEXPORT void s_matrix_multiply(const cublasHandle_t blasHandle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const float alpha, const float x[], const float y[], const float beta, float c[]){ + int lda = transA == CUBLAS_OP_N ? m : k; + int ldb = transB == CUBLAS_OP_N ? k : n; + + cuda_gemm(blasHandle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m, cublasSgemm); + } + + DLLEXPORT void d_matrix_multiply(const cublasHandle_t blasHandle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const double alpha, const double x[], const double y[], const double beta, double c[]){ + int lda = transA == CUBLAS_OP_N ? m : k; + int ldb = transB == CUBLAS_OP_N ? k : n; + + cuda_gemm(blasHandle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m, cublasDgemm); + } + + DLLEXPORT void c_matrix_multiply(const cublasHandle_t blasHandle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const cuComplex alpha, const cuComplex x[], const cuComplex y[], const cuComplex beta, cuComplex c[]){ + int lda = transA == CUBLAS_OP_N ? m : k; + int ldb = transB == CUBLAS_OP_N ? k : n; + + cuda_gemm(blasHandle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m, cublasCgemm); + } + + DLLEXPORT void z_matrix_multiply(const cublasHandle_t blasHandle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const cuDoubleComplex alpha, const cuDoubleComplex x[], const cuDoubleComplex y[], const cuDoubleComplex beta, cuDoubleComplex c[]){ + int lda = transA == CUBLAS_OP_N ? m : k; + int ldb = transB == CUBLAS_OP_N ? k : n; + + cuda_gemm(blasHandle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m, cublasZgemm); + } + +#if GCC +} +#endif diff --git a/src/NativeProviders/CUDA/lapack.cpp b/src/NativeProviders/CUDA/lapack.cpp index 189a0b4c..7798df12 100644 --- a/src/NativeProviders/CUDA/lapack.cpp +++ b/src/NativeProviders/CUDA/lapack.cpp @@ -1,521 +1,976 @@ +#include + #include "lapack_common.h" #include "wrapper_common.h" -#include "cublas.h" +#include "cublas_v2.h" #include "cusolverDn.h" -#include +#include "cuda_runtime.h" -template -inline int lu_factor(int m, T a[], int ipiv[], - int(*getrf) (CBLAS_ORDER, const int, const int, K*, const int, int*)) +template +inline int lu_factor(cusolverDnHandle_t solverHandle, int m, T a[], int ipiv[], GETRF getrf, GETRFBSIZE getrfbsize) { - int info = getrf(CblasColMajor, m, m, a, m, ipiv); + int info = 0; + T* work = NULL; + int lwork = 0; + + T* d_A = NULL; + cudaMalloc((void**)&d_A, m*m*sizeof(T)); + cublasSetMatrix(m, m, sizeof(T), a, m, d_A, m); + + int* d_I = NULL; + cudaMalloc((void**)&d_I, m*sizeof(int)); + + getrfbsize(solverHandle, m, m, a, m, &lwork); + cudaMalloc((void**)lwork, sizeof(T)*lwork); + + getrf(solverHandle, m, m, d_A, m, work, d_I, &info); + + cublasGetMatrix(m, m, sizeof(T), d_A, m, a, m); + cublasGetVector(m, sizeof(T), d_I, 1, ipiv, 1); + shift_ipiv_down(m, ipiv); + + cudaFree(d_A); + cudaFree(d_I); + cudaFree(work); + return info; }; -template -inline int lu_inverse(int n, T a[], - int(*getrf) (CBLAS_ORDER, const int, const int, K*, const int, int*), - int(*getri) (CBLAS_ORDER, const int, K*, const int, const int*)) +template +inline int lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int n, T a[], GETRF getrf, GETRI getri, GETRFBSIZE getrfbsize) { - int* ipiv = new int[n]; - int info = getrf(CblasColMajor, n, n, a, n, ipiv); + int info = 0; + T* work = NULL; + int lwork = 0; + + int* d_I = NULL; + cudaMalloc((void**)&d_I, n*sizeof(T)); + + T* d_A = NULL; + cudaMalloc((void**)&d_A, n*n*sizeof(T)); + cublasSetMatrix(n, n, sizeof(T), a, n, d_A, n); - if (info != 0){ - delete[] ipiv; + getrfbsize(solverHandle, n, n, d_A, n, &lwork); + cudaMalloc((void**)lwork, sizeof(T)*lwork); + + getrf(solverHandle, n, n, d_A, n, work, d_I, &info); + + cudaFree(work); + + if (info != 0) + { + cudaFree(d_A); + cudaFree(d_I); return info; } - info = getri(CblasColMajor, n, a, n, ipiv); - delete[] ipiv; + T* d_C = NULL; + cudaMalloc((void**)&d_C, n*n*sizeof(T)); + + getri(blasHandle, n, d_A, n, d_I, d_C, n, &info); + + cublasGetMatrix(n, n, sizeof(T), d_C, n, a, n); + + cudaFree(d_A); + cudaFree(d_I); + cudaFree(d_C); + return info; }; -template -inline int lu_inverse_factored(int n, T a[], int ipiv[], - int(*getri) (CBLAS_ORDER, const int, K*, const int, const int*)) +template +inline int lu_inverse_factored(cublasHandle_t blasHandle, int n, T a[], int ipiv[], GETRI getri) { shift_ipiv_up(n, ipiv); - int info = getri(CblasColMajor, n, a, n, ipiv); + int info = 0; + + T* d_A = NULL; + cudaMalloc((void**)&d_A, n*n*sizeof(T)); + cublasSetMatrix(n, n, sizeof(T), a, n, d_A, n); + + T* d_C = NULL; + cudaMalloc((void**)&d_C, n*n*sizeof(T)); + + int* d_I = NULL; + cudaMalloc((void**)&d_I, n*sizeof(int)); + cublasSetVector(n, sizeof(int), ipiv, 1, d_I, 1); + + getri(blasHandle, n, d_A, n, d_I, d_C, n, &info); + + cublasGetMatrix(n, n, sizeof(T), d_C, n, a, n); + cublasGetVector(n, sizeof(int), d_I, 1, ipiv, 1); + shift_ipiv_down(n, ipiv); + + cudaFree(d_A); + cudaFree(d_I); + cudaFree(d_C); + return info; } -template -inline int lu_solve_factored(int n, int nrhs, T a[], int ipiv[], T b[], - int(*getrs) (CBLAS_ORDER, CBLAS_TRANSPOSE, const int, const int, const K*, const int, const int*, K*, const int)) +template +inline int lu_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[], int ipiv[], T b[], GETRS getrs) { shift_ipiv_up(n, ipiv); - int info = getrs(CblasColMajor, CblasNoTrans, n, nrhs, a, n, ipiv, b, n); + int info = 0; + + T* d_A = NULL; + cudaMalloc((void**)&d_A, n*n*sizeof(T)); + cublasSetMatrix(n, n, sizeof(T), a, n, d_A, n); + + T* d_B = NULL; + cudaMalloc((void**)&d_B, n*nrhs*sizeof(T)); + cublasSetMatrix(n, nrhs, sizeof(T), b, n, d_B, n); + + int* d_I = NULL; + cudaMalloc((void**)&d_I, n*sizeof(int)); + cublasSetVector(n, sizeof(int), ipiv, 1, d_I, 1); + + getrs(solverHandle, CUBLAS_OP_N, n, nrhs, d_A, n, d_I, d_B, n, &info); + + cublasGetMatrix(n, nrhs, sizeof(T), d_B, n, b, n); + shift_ipiv_down(n, ipiv); + + cudaFree(d_A); + cudaFree(d_B); + cudaFree(d_I); + return info; } -template -inline int lu_solve(int n, int nrhs, T a[], T b[], - int(*getrf) (CBLAS_ORDER, const int, const int, K*, const int, int*), - int(*getrs) (CBLAS_ORDER, CBLAS_TRANSPOSE, const int, const int, const K*, const int, const int*, K*, const int)) +template +inline int lu_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[], T b[], GETRF getrf, GETRS getrs, GETRFBSIZE getrfbsize) { - T* clone = Clone(n, n, a); - int* ipiv = new int[n]; - int info = getrf(CblasColMajor, n, n, clone, n, ipiv); + int info = 0; + T* work = NULL; + int lwork = 0; + + int* d_I = NULL; + cudaMalloc((void**)&d_I, n*sizeof(T)); + + T* d_A = NULL; + cudaMalloc((void**)&d_A, n*n*sizeof(T)); + cublasSetMatrix(n, n, sizeof(T), a, n, d_A, n); + + getrfbsize(solverHandle, n, n, a, n, &lwork); + cudaMalloc((void**)lwork, sizeof(T)*lwork); + + getrf(solverHandle, n, n, d_A, n, work, d_I, &info); - if (info != 0){ - delete[] ipiv; - delete[] clone; + if (info != 0) + { + cudaFree(d_I); + cudaFree(d_A); return info; } - info = getrs(CblasColMajor, CblasNoTrans, n, nrhs, clone, n, ipiv, b, n); - delete[] ipiv; - delete[] clone; + T* d_B = NULL; + cudaMalloc((void**)&d_B, n*nrhs*sizeof(T)); + cublasSetMatrix(n, nrhs, sizeof(T), b, n, d_B, n); + + getrs(solverHandle, CUBLAS_OP_N, n, nrhs, d_A, n, d_I, d_B, n, &info); + + cublasGetMatrix(n, nrhs, sizeof(T), d_B, n, b, n); + + cudaFree(d_A); + cudaFree(d_B); + cudaFree(d_I); + return info; } -template -inline int cholesky_factor(int n, T* a, int(*potrf) (CBLAS_ORDER, CBLAS_UPLO, const int, K*, const int)) + +template +inline int cholesky_factor(cusolverDnHandle_t solverHandle, int n, T* a, POTRF potrf, POTRFBSIZE potrfbsize) { - int info = potrf(CblasColMajor, CblasLower, n, a, n); + int info = 0; + + T* d_A = NULL; + cudaMalloc((void**)&d_A, n*n*sizeof(T)); + cublasSetMatrix(n, n, sizeof(T), a, n, d_A, n); + + T* work = NULL; + int lWork = 0; + potrfbsize(solverHandle, CUBLAS_FILL_MODE_LOWER, n, d_A, n, &lWork); + cudaMalloc((void**)&work, sizeof(T)*lWork); + + potrf(solverHandle, CUBLAS_FILL_MODE_LOWER, n, d_A, n, work, lWork, &info); + + cublasGetMatrix(n, n, sizeof(T), d_A, n, a, n); + T zero = T(); + for (int i = 0; i < n; ++i) { int index = i * n; + for (int j = 0; j < n && i > j; ++j) { a[index + j] = zero; } } - return info; -} - -template -inline int cholesky_solve(int n, int nrhs, T a[], T b[], - int(*potrf) (CBLAS_ORDER, CBLAS_UPLO, const int, K*, const int), - int(*potrs) (CBLAS_ORDER, CBLAS_UPLO, const int, const int, const K*, const int, K*, const int)) -{ - T* clone = Clone(n, n, a); - int info = potrf(CblasColMajor, CblasLower, n, clone, n); - if (info != 0){ - delete[] clone; - return info; - } + cudaFree(d_A); + cudaFree(work); - info = potrs(CblasColMajor, CblasLower, n, nrhs, clone, n, b, n); - delete[] clone; return info; } -template -inline int cholesky_solve_factored(int n, int nrhs, T a[], T b[], - int(*potrs) (CBLAS_ORDER, CBLAS_UPLO, const int, const int, const K*, const int, K*, const int)) +template +inline int cholesky_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[], T b[], POTRF potrf, POTRS potrs, POTRFBSIZE potrfbsize) { - return potrs(CblasColMajor, CblasLower, n, nrhs, a, n, b, n); -} + int info; -template -inline int qr_factor(int m, int n, T r[], T tau[], T q[], T work[], int len, - int(*geqrf) (const int, const int, K*, const int, T*), - int(*orgqr) (const int, const int, const int, K*, const int, const K*)) -{ - int info = geqrf(m, n, r, m, tau); + T* d_A = NULL; + cudaMalloc((void**)&d_A, n*n*sizeof(T)); + cublasSetMatrix(n, n, sizeof(T), a, n, d_A, n); - for (int i = 0; i < m; ++i) - { - for (int j = 0; j < m && j < n; ++j) - { - if (i > j) - { - q[j * m + i] = r[j * m + i]; - } - } - } + T* work = NULL; + int lWork = 0; + potrfbsize(solverHandle, CUBLAS_FILL_MODE_LOWER, n, d_A, n, &lWork); + cudaMalloc((void**)&work, sizeof(T)*lWork); - //compute the q elements explicitly - if (m <= n) - { - info = orgqr(m, m, m, q, m, tau); - } - else + potrf(solverHandle, CUBLAS_FILL_MODE_LOWER, n, d_A, n, work, lWork, &info); + + cudaFree(work); + + if (info != 0) { - info = orgqr(m, m, n, q, m, tau); + cudaFree(d_A); + return info; } - return info; -} + T* d_B = NULL; + cudaMalloc((void**)d_B, n*nrhs*sizeof(T)); + cublasSetMatrix(n, nrhs, sizeof(T), b, n, d_B, n); -template -inline int qr_thin_factor(int m, int n, T q[], T tau[], T r[], T work[], int len, - void(*geqrf) (const int*, const int*, T*, const int*, T*, T*, const int*, int*), - void(*orgqr) (const int*, const int*, const int*, T*, const int*, const T*, T*, const int*, int*)) -{ - int info = 0; - geqrf(&m, &n, q, &m, tau, work, &len, &info); + potrs(solverHandle, CUBLAS_FILL_MODE_LOWER, n, nrhs, d_A, n, d_B, n, &info); - for (int i = 0; i < n; ++i) - { - for (int j = 0; j < n; ++j) - { - if (i <= j) { - r[j * n + i] = q[j * m + i]; - } - } - } + cublasGetMatrix(n, nrhs, sizeof(T), d_B, n, b, n); - orgqr(&m, &n, &n, q, &m, tau, work, &len, &info); + cudaFree(d_A); + cudaFree(d_B); return info; } -template -inline int qr_solve(int m, int n, int bn, T a[], T b[], T x[], T work[], int len, - void(*gels) (const char*, const int*, const int*, const int*, T*, - const int*, T* b, const int*, T*, const int*, int*)) +template +inline int cholesky_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[], T b[], POTRS potrs) { - T* clone_a = new T[m*n]; - std::memcpy(clone_a, a, m*n*sizeof(T)); + int info; - T* clone_b = new T[m*bn]; - std::memcpy(clone_b, b, m*bn*sizeof(T)); + T* d_A = NULL; + cudaMalloc((void**)&d_A, n*n*sizeof(T)); + cublasSetMatrix(n, n, sizeof(T), a, n, d_A, n); - char N = 'N'; - int info = 0; - gels(&N, &m, &n, &bn, clone_a, &m, clone_b, &m, work, &len, &info); - copyBtoX(n, n, bn, clone_b, x); + T* d_B = NULL; + cudaMalloc((void**)d_B, n*nrhs*sizeof(T)); + cublasSetMatrix(n, nrhs, sizeof(T), b, n, d_B, n); - delete[] clone_a; - delete[] clone_b; - return info; -} + potrs(solverHandle, CUBLAS_FILL_MODE_LOWER, n, nrhs, d_A, n, d_B, n, &info); -template -inline int qr_solve_factored(int m, int n, int bn, T r[], T b[], T tau[], T x[], T work[], int len, - void(*ormqr) (const char*, const char*, const int*, const int*, const int*, - const T*, const int*, const T*, T*, const int*, T*, const int*, int* info), - void(*trsm) (const CBLAS_ORDER, const CBLAS_SIDE, const CBLAS_UPLO, const CBLAS_TRANSPOSE, const CBLAS_DIAG, - const int, const int, const T, const T*, const int, T*, const int)) -{ - T* clone_b = new T[m*bn]; - std::memcpy(clone_b, b, m*bn*sizeof(T)); + cublasGetMatrix(n, nrhs, sizeof(T), d_B, n, b, n); - char side = 'L'; - char tran = 'T'; - int info = 0; - ormqr(&side, &tran, &m, &bn, &n, r, &m, tau, clone_b, &m, work, &len, &info); - trsm(CblasColMajor, CblasLeft, CblasUpper, CblasNoTrans, CblasNonUnit, n, bn, 1.0, r, m, clone_b, m); - copyBtoX(n, n, bn, clone_b, x); + cudaFree(d_A); + cudaFree(d_B); - delete[] clone_b; return info; } -template -inline int complex_qr_solve_factored(int m, int n, int bn, T r[], T b[], T tau[], T x[], T work[], int len, - void(*unmqr) (const char*, const char*, const int*, const int*, const int*, - const T*, const int*, const T*, T*, const int*, T*, const int*, int* info), - void(*trsm) (const CBLAS_ORDER, const CBLAS_SIDE, const CBLAS_UPLO, const CBLAS_TRANSPOSE, const CBLAS_DIAG, - const int, const int, const void*, const void*, const int, void*, const int ldb)) +//template +//inline int qr_factor(int m, int n, T r[], T tau[], T q[], T work[], int len, GEQRF geqrf, ORGQR orgqr) +//{ +// int info = 0; +// geqrf(&m, &n, r, &m, tau, work, &len, &info); +// +// for (int i = 0; i < m; ++i) +// { +// for (int j = 0; j < m && j < n; ++j) +// { +// if (i > j) +// { +// q[j * m + i] = r[j * m + i]; +// } +// } +// } +// +// //compute the q elements explicitly +// if (m <= n) +// { +// orgqr(&m, &m, &m, q, &m, tau, work, &len, &info); +// } +// else +// { +// orgqr(&m, &m, &n, q, &m, tau, work, &len, &info); +// } +// +// return info; +//} +// +//template +//inline int qr_thin_factor(int m, int n, T q[], T tau[], T r[], T work[], int len, GEQRF geqrf, ORGQR orgqr) +//{ +// int info = 0; +// geqrf(&m, &n, q, &m, tau, work, &len, &info); +// +// for (int i = 0; i < n; ++i) +// { +// for (int j = 0; j < n; ++j) +// { +// if (i <= j) +// { +// r[j * n + i] = q[j * m + i]; +// } +// } +// } +// +// orgqr(&m, &n, &n, q, &m, tau, work, &len, &info); +// return info; +//} +// +//template +//inline int qr_solve(int m, int n, int bn, T a[], T b[], T x[], T work[], int len, GELS gels) +//{ +// T* clone_a = Clone(m, n, a); +// T* clone_b = Clone(m, bn, b); +// char N = 'N'; +// int info = 0; +// gels(&N, &m, &n, &bn, clone_a, &m, clone_b, &m, work, &len, &info); +// copyBtoX(m, n, bn, clone_b, x); +// delete[] clone_a; +// delete[] clone_b; +// return info; +//} + +//template +//inline int qr_solve_factored(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int m, int n, int bn, T r[], T b[], T tau[], T x[], T work[], int len, ORMQR ormqr, TRSM trsm) +//{ +// T* clone_b = Clone(m, bn, b); +// char side = 'L'; +// char tran = 'T'; +// int info = 0; +// ormqr(solverHandle, &side, &tran, &m, &bn, &n, r, &m, tau, clone_b, &m, work, &len, &info); +// trsm(blasHandle, CblasColMajor, CblasLeft, CblasUpper, CblasNoTrans, CblasNonUnit, n, bn, 1.0, r, m, clone_b, m); +// +// copyBtoX(m, n, bn, clone_b, x); +// delete[] clone_b; +// return info; +//} + +//template +//inline int complex_qr_solve_factored(int m, int n, int bn, T r[], T b[], T tau[], T x[], T work[], int len, UNMQR unmqr, TRSM trsm) +//{ +// T* clone_b = Clone(m, bn, b); +// char side = 'L'; +// char tran = 'C'; +// int info = 0; +// unmqr(&side, &tran, &m, &bn, &n, r, &m, tau, clone_b, &m, work, &len, &info); +// T one = 1.0f; +// trsm(CblasColMajor, CblasLeft, CblasUpper, CblasNoTrans, CblasNonUnit, n, bn, &one, r, m, clone_b, m); +// copyBtoX(m, n, bn, clone_b, x); +// delete[] clone_b; +// return info; +//} + +template +inline int svd_factor(cusolverDnHandle_t solverHandle, bool compute_vectors, int m, int n, T a[], T s[], T u[], T v[], GESVD gesvd, GESVDBSIZE gesvdbsize) { - T* clone_b = new T[m*bn]; - std::memcpy(clone_b, b, m*bn*sizeof(T)); - - char side = 'L'; - char tran = 'C'; int info = 0; - unmqr(&side, &tran, &m, &bn, &n, r, &m, tau, clone_b, &m, work, &len, &info); + int dim_s = std::min(m, n); - T one = { 1.0f, 0.0f }; - trsm(CblasColMajor, CblasLeft, CblasUpper, CblasNoTrans, CblasNonUnit, n, bn, &one, r, m, clone_b, m); - copyBtoX(n, n, bn, clone_b, x); + T* d_A = NULL; + cudaMalloc((void**)&d_A, m*n*sizeof(T)); + cublasSetMatrix(m, n, sizeof(T), a, m, d_A, m); - delete[] clone_b; - return info; -} + T* d_S = NULL; + cudaMalloc((void**)&d_S, dim_s*sizeof(T)); + + T* d_U = NULL; + cudaMalloc((void**)&d_U, m*m*sizeof(T)); + + T* d_V = NULL; + cudaMalloc((void**)&d_V, n*m*sizeof(T)); + + T* work = NULL; + int lWork = 0; + gesvdbsize(solverHandle, m, n, &lWork); + cudaMalloc((void**)&work, lWork*sizeof(T)); + + T* rwork = NULL; + cudaMalloc((void**)&rwork, 5 * dim_s * sizeof(T)); -template -inline int svd_factor(bool compute_vectors, int m, int n, T a[], T s[], T u[], T v[], T work[], int len, - void(*gesvd) (const char*, const char*, const int*, const int*, T*, const int*, - T*, T*, const int*, T*, const int*, T*, const int*, int*)) -{ - int info = 0; char job = compute_vectors ? 'A' : 'N'; - gesvd(&job, &job, &m, &n, a, &m, s, u, &m, v, &n, work, &len, &info); + gesvd(solverHandle, job, job, m, n, d_A, m, d_S, d_U, m, d_V, n, work, lWork, rwork, &info); + + cublasGetVector(dim_s, sizeof(T), d_S, 1, s, 1); + cublasGetMatrix(m, m, sizeof(T), d_U, m, u, m); + cublasGetMatrix(n, n, sizeof(T), d_V, n, v, n); + + cudaFree(d_A); + cudaFree(d_S); + cudaFree(d_U); + cudaFree(d_V); + cudaFree(work); + cudaFree(rwork); + return info; } - -template -inline int complex_svd_factor(bool compute_vectors, int m, int n, T a[], T s[], T u[], T v[], T work[], int len, - void(*gesvd) (const char*, const char*, const int*, const int*, T*, const int*, - R*, T*, const int*, T*, const int*, T*, const int*, R*, int*)) +template +inline int complex_svd_factor(cusolverDnHandle_t solverHandle, bool compute_vectors, int m, int n, T a[], T s[], T u[], T v[], GESVD gesvd, GESVDBSIZE gesvdbsize) { int info = 0; int dim_s = std::min(m, n); - R* rwork = new R[5 * dim_s]; + + T* d_A = NULL; + cudaMalloc((void**)&d_A, m*n*sizeof(T)); + cublasSetMatrix(m, n, sizeof(T), a, m, d_A, m); + R* s_local = new R[dim_s]; + R* d_S = NULL; + cudaMalloc((void**)&d_S, dim_s*sizeof(R)); + + T* d_U = NULL; + cudaMalloc((void**)&d_U, m*m*sizeof(T)); + + T* d_V = NULL; + cudaMalloc((void**)&d_V, n*m*sizeof(T)); + + T* work = NULL; + int lWork = 0; + gesvdbsize(solverHandle, m, n, &lWork); + cudaMalloc((void**)&work, lWork*sizeof(T)); + + R* rwork = NULL; + cudaMalloc((void**)&rwork, 5 * dim_s * sizeof(R)); + char job = compute_vectors ? 'A' : 'N'; - gesvd(&job, &job, &m, &n, a, &m, s_local, u, &m, v, &n, work, &len, rwork, &info); + gesvd(solverHandle, job, job, m, n, d_A, m, d_S, d_U, m, d_V, n, work, lWork, rwork, &info); + + cublasGetVector(dim_s, sizeof(T), d_S, 1, s_local, 1); + cublasGetMatrix(m, m, sizeof(T), d_U, m, u, m); + cublasGetMatrix(n, n, sizeof(T), d_V, n, v, n); - for (int index = 0; index < dim_s; ++index){ - T value = { s_local[index], 0.0f }; - s[index] = value; + for (int index = 0; index < dim_s; ++index) + { + s[index].x = s_local[index]; } - delete[] rwork; delete[] s_local; + cudaFree(d_A); + cudaFree(d_S); + cudaFree(d_U); + cudaFree(d_V); + cudaFree(work); + cudaFree(rwork); + return info; } +//template +//inline int eigen_factor(int n, T a[], T vectors[], R values[], T d[], GEES gees, TREVC trevc) +//{ +// T* clone_a = Clone(n, n, a); +// T* wr = new T[n]; +// T* wi = new T[n]; +// +// int sdim; +// int info = gees(LAPACK_COL_MAJOR, 'V', 'N', nullptr, n, clone_a, n, &sdim, wr, wi, vectors, n); +// if (info != 0) +// { +// delete[] clone_a; +// delete[] wr; +// delete[] wi; +// return info; +// } +// +// int m; +// info = trevc(LAPACK_COL_MAJOR, 'R', 'B', nullptr, n, clone_a, n, nullptr, n, vectors, n, n, &m); +// if (info != 0) +// { +// delete[] clone_a; +// delete[] wr; +// delete[] wi; +// return info; +// } +// +// for (int index = 0; index < n; ++index) +// { +// values[index] = R(wr[index], wi[index]); +// } +// +// for (int i = 0; i < n; ++i) +// { +// int in = i * n; +// d[in + i] = wr[i]; +// +// if (wi[i] > 0) +// { +// d[in + n + i] = wi[i]; +// } +// else if (wi[i] < 0) +// { +// d[in - n + i] = wi[i]; +// } +// } +// +// delete[] clone_a; +// delete[] wr; +// delete[] wi; +// return info; +//} +// +//template +//inline int eigen_complex_factor(int n, T a[], T vectors[], cuDoubleComplex values[], T d[], GEES gees, TREVC trevc) +//{ +// T* clone_a = Clone(n, n, a); +// T* w = new T[n]; +// +// int sdim; +// int info = gees(LAPACK_COL_MAJOR, 'V', 'N', nullptr, n, clone_a, n, &sdim, w, vectors, n); +// if (info != 0) +// { +// delete[] clone_a; +// delete[] w; +// return info; +// } +// +// int m; +// info = trevc(LAPACK_COL_MAJOR, 'R', 'B', nullptr, n, clone_a, n, nullptr, n, vectors, n, n, &m); +// if (info != 0) +// { +// delete[] clone_a; +// delete[] w; +// return info; +// } +// +// for (int i = 0; i < n; ++i) +// { +// values[i] = w[i]; +// d[i * n + i] = w[i]; +// } +// +// delete[] clone_a; +// delete[] w; +// return info; +//} +// +//template +//inline int sym_eigen_factor(int n, T a[], T vectors[], cuDoubleComplex values[], T d[], SYEV syev) +//{ +// T* clone_a = Clone(n, n, a); +// R* w = new R[n]; +// +// int info = syev(LAPACK_COL_MAJOR, 'V', 'U', n, clone_a, n, w); +// if (info != 0) +// { +// delete[] clone_a; +// delete[] w; +// return info; +// } +// +// memcpy(vectors, clone_a, n*n*sizeof(T)); +// +// for (int index = 0; index < n; ++index) +// { +// values[index] = cuDoubleComplex(w[index]); +// } +// +// for (int j = 0; j < n; ++j) +// { +// int jn = j*n; +// +// for (int i = 0; i < n; ++i) +// { +// if (i == j) +// { +// d[jn + i] = w[i]; +// } +// } +// } +// +// delete[] clone_a; +// delete[] w; +// return info; +//} + +#define sgetrf cusolverDnSgetrf +#define dgetrf cusolverDnDgetrf +#define cgetrf cusolverDnCgetrf +#define zgetrf cusolverDnZgetrf +#define sgetrfbsize cusolverDnSgetrf_bufferSize +#define dgetrfbsize cusolverDnDgetrf_bufferSize +#define cgetrfbsize cusolverDnCgetrf_bufferSize +#define zgetrfbsize cusolverDnZgetrf_bufferSize + +#define sgetrs cusolverDnSgetrs +#define dgetrs cusolverDnDgetrs +#define cgetrs cusolverDnCgetrs +#define zgetrs cusolverDnZgetrs + +#define spotrf cusolverDnSpotrf +#define dpotrf cusolverDnDpotrf +#define cpotrf cusolverDnCpotrf +#define zpotrf cusolverDnZpotrf +#define spotrfbsize cusolverDnSpotrf_bufferSize +#define dpotrfbsize cusolverDnDpotrf_bufferSize +#define cpotrfbsize cusolverDnCpotrf_bufferSize +#define zpotrfbsize cusolverDnZpotrf_bufferSize + +#define spotrs cusolverDnSpotrs +#define dpotrs cusolverDnDpotrs +#define cpotrs cusolverDnCpotrs +#define zpotrs cusolverDnZpotrs + +#define sgeqrf cusolverDnSgeqrf +#define dgeqrf cusolverDnDgeqrf +#define cgeqrf cusolverDnCgeqrf +#define zgeqrf cusolverDnZgeqrf + +#define sormqr cusolverDnSormqr +#define dormqr cusolverDnDormqr + +#define sgesvd cusolverDnSgesvd +#define dgesvd cusolverDnDgesvd +#define cgesvd cusolverDnCgesvd +#define zgesvd cusolverDnZgesvd +#define sgesvdbsize cusolverDnSgesvd_bufferSize +#define dgesvdbsize cusolverDnDgesvd_bufferSize +#define cgesvdbsize cusolverDnCgesvd_bufferSize +#define zgesvdbsize cusolverDnZgesvd_bufferSize + + +inline int sgetri(cublasHandle_t handle, int n, const float a[], int lda, const int *ipiv, float c[], int ldc, int *info) +{ + return cublasSgetriBatched(handle, n, &a, lda, ipiv, &c, ldc, info, 1); +} + +inline int dgetri(cublasHandle_t handle, int n, const double a[], int lda, const int *ipiv, double c[], int ldc, int *info) +{ + return cublasDgetriBatched(handle, n, &a, lda, ipiv, &c, ldc, info, 1); +} + +inline int cgetri(cublasHandle_t handle, int n, const cuComplex a[], int lda, const int *ipiv, cuComplex c[], int ldc, int *info) +{ + return cublasCgetriBatched(handle, n, &a, lda, ipiv, &c, ldc, info, 1); +} + +inline int zgetri(cublasHandle_t handle, int n, const cuDoubleComplex a[], int lda, const int *ipiv, cuDoubleComplex c[], int ldc, int *info) +{ + return cublasZgetriBatched(handle, n, &a, lda, ipiv, &c, ldc, info, 1); +} + extern "C" { - DLLEXPORT int s_lu_factor(int m, float a[], int ipiv[]) { - return lu_factor(m, a, ipiv, cusolverDnSgetrf); + + DLLEXPORT int s_lu_factor(cusolverDnHandle_t solverHandle, int m, float a[], int ipiv[]) + { + return lu_factor(solverHandle, m, a, ipiv, sgetrf, sgetrfbsize); } - DLLEXPORT int d_lu_factor(int m, double a[], int ipiv[]) { - return lu_factor(m, a, ipiv, cusolverDnDgetrf); + DLLEXPORT int d_lu_factor(cusolverDnHandle_t solverHandle, int m, double a[], int ipiv[]) + { + return lu_factor(solverHandle, m, a, ipiv, dgetrf, dgetrfbsize); } - DLLEXPORT int c_lu_factor(int m, cuComplex a[], int ipiv[]) { - return lu_factor(m, a, ipiv, cusolverDnCgetrf); + DLLEXPORT int c_lu_factor(cusolverDnHandle_t solverHandle, int m, cuComplex a[], int ipiv[]) + { + return lu_factor(solverHandle, m, a, ipiv, cgetrf, cgetrfbsize); } - DLLEXPORT int z_lu_factor(int m, cuDoubleComplex a[], int ipiv[]) { - return lu_factor(m, a, ipiv, cusolverDnZgetrf); + DLLEXPORT int z_lu_factor(cusolverDnHandle_t solverHandle, int m, cuDoubleComplex a[], int ipiv[]) + { + return lu_factor(solverHandle, m, a, ipiv, zgetrf, zgetrfbsize); } - DLLEXPORT int s_lu_inverse(int n, float a[]) + DLLEXPORT int s_lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int n, float a[]) { - return lu_inverse(n, a, cusolverDnSgetrf, cusolverDnSgetri); + return lu_inverse(solverHandle, blasHandle, n, a, sgetrf, sgetri, sgetrfbsize); } - DLLEXPORT int d_lu_inverse(int n, double a[]) + DLLEXPORT int d_lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int n, double a[]) { - return lu_inverse(n, a, cusolverDnDgetrf, cusolverDnDgetri); + return lu_inverse(solverHandle, blasHandle, n, a, dgetrf, dgetri, dgetrfbsize); } - DLLEXPORT int c_lu_inverse(int n, cuComplex a[]) + DLLEXPORT int c_lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int n, cuComplex a[]) { - return lu_inverse(n, a, cusolverDnCgetrf, cusolverDnCgetri); + return lu_inverse(solverHandle, blasHandle, n, a, cgetrf, cgetri, cgetrfbsize); } - DLLEXPORT int z_lu_inverse(int n, cuDoubleComplex a[]) + DLLEXPORT int z_lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int n, cuDoubleComplex a[]) { - return lu_inverse(n, a, cusolverDnZgetrf, cusolverDnZgetri); + return lu_inverse(solverHandle, blasHandle, n, a, zgetrf, zgetri, zgetrfbsize); } - DLLEXPORT int s_lu_inverse_factored(int n, float a[], int ipiv[], float work[], int lwork) + DLLEXPORT int s_lu_inverse_factored(cublasHandle_t blasHandle, int n, float a[], int ipiv[]) { - return lu_inverse_factored(n, a, ipiv, cusolverDnSgetri); + return lu_inverse_factored(blasHandle, n, a, ipiv, sgetri); } - DLLEXPORT int d_lu_inverse_factored(int n, double a[], int ipiv[], double work[], int lwork) + DLLEXPORT int d_lu_inverse_factored(cublasHandle_t blasHandle, int n, double a[], int ipiv[]) { - return lu_inverse_factored(n, a, ipiv, cusolverDnDgetri); + return lu_inverse_factored(blasHandle, n, a, ipiv, dgetri); } - DLLEXPORT int c_lu_inverse_factored(int n, cuComplex a[], int ipiv[], cuComplex work[], int lwork) + DLLEXPORT int c_lu_inverse_factored(cublasHandle_t blasHandle, int n, cuComplex a[], int ipiv[]) { - return lu_inverse_factored(n, a, ipiv, cusolverDnCgetri); + return lu_inverse_factored(blasHandle, n, a, ipiv, cgetri); } - DLLEXPORT int z_lu_inverse_factored(int n, cuDoubleComplex a[], int ipiv[], cuDoubleComplex work[], int lwork) + DLLEXPORT int z_lu_inverse_factored(cublasHandle_t blasHandle, int n, cuDoubleComplex a[], int ipiv[]) { - return lu_inverse_factored(n, a, ipiv, cusolverDnZgetri); + return lu_inverse_factored(blasHandle, n, a, ipiv, zgetri); } - DLLEXPORT int s_lu_solve_factored(int n, int nrhs, float a[], int ipiv[], float b[]) + DLLEXPORT int s_lu_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, float a[], int ipiv[], float b[]) { - return lu_solve_factored(n, nrhs, a, ipiv, b, cusolverDnSgetrs); + return lu_solve_factored(solverHandle, n, nrhs, a, ipiv, b, sgetrs); } - DLLEXPORT int d_lu_solve_factored(int n, int nrhs, double a[], int ipiv[], double b[]) + DLLEXPORT int d_lu_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, double a[], int ipiv[], double b[]) { - return lu_solve_factored(n, nrhs, a, ipiv, b, cusolverDnDgetrs); + return lu_solve_factored(solverHandle, n, nrhs, a, ipiv, b, dgetrs); } - DLLEXPORT int c_lu_solve_factored(int n, int nrhs, cuComplex a[], int ipiv[], cuComplex b[]) + DLLEXPORT int c_lu_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, cuComplex a[], int ipiv[], cuComplex b[]) { - return lu_solve_factored(n, nrhs, a, ipiv, b, cusolverDnCgetrs); + return lu_solve_factored(solverHandle, n, nrhs, a, ipiv, b, cgetrs); } - DLLEXPORT int z_lu_solve_factored(int n, int nrhs, cuDoubleComplex a[], int ipiv[], cuDoubleComplex b[]) + DLLEXPORT int z_lu_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, cuDoubleComplex a[], int ipiv[], cuDoubleComplex b[]) { - return lu_solve_factored(n, nrhs, a, ipiv, b, cusolverDnZgetrs); + return lu_solve_factored(solverHandle, n, nrhs, a, ipiv, b, zgetrs); } - DLLEXPORT int s_lu_solve(int n, int nrhs, float a[], float b[]) + DLLEXPORT int s_lu_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, float a[], float b[]) { - return lu_solve(n, nrhs, a, b, cusolverDnSgetrf, cusolverDnSgetrs); + return lu_solve(solverHandle, n, nrhs, a, b, sgetrf, sgetrs, sgetrfbsize); } - DLLEXPORT int d_lu_solve(int n, int nrhs, double a[], double b[]) + DLLEXPORT int d_lu_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, double a[], double b[]) { - return lu_solve(n, nrhs, a, b, cusolverDnDgetrf, cusolverDnDgetrs); + return lu_solve(solverHandle, n, nrhs, a, b, dgetrf, dgetrs, dgetrfbsize); } - DLLEXPORT int c_lu_solve(int n, int nrhs, cuComplex a[], cuComplex b[]) + DLLEXPORT int c_lu_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, cuComplex a[], cuComplex b[]) { - return lu_solve(n, nrhs, a, b, cusolverDnCgetrf, cusolverDnCgetrs); + return lu_solve(solverHandle, n, nrhs, a, b, cgetrf, cgetrs, cgetrfbsize); } - DLLEXPORT int z_lu_solve(int n, int nrhs, cuDoubleComplex a[], cuDoubleComplex b[]) + DLLEXPORT int z_lu_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, cuDoubleComplex a[], cuDoubleComplex b[]) { - return lu_solve(n, nrhs, a, b, cusolverDnZgetrf, cusolverDnZgetrs); + return lu_solve(solverHandle, n, nrhs, a, b, zgetrf, zgetrs, zgetrfbsize); } - DLLEXPORT int s_cholesky_factor(int n, float a[]){ - return cholesky_factor(n, a, cusolverDnSpotrf); + DLLEXPORT int s_cholesky_factor(cusolverDnHandle_t solverHandle, int n, float a[]) + { + return cholesky_factor(solverHandle, n, a, spotrf, spotrfbsize); } - DLLEXPORT int d_cholesky_factor(int n, double* a){ - return cholesky_factor(n, a, cusolverDnDpotrf); + DLLEXPORT int d_cholesky_factor(cusolverDnHandle_t solverHandle, int n, double* a) + { + return cholesky_factor(solverHandle, n, a, dpotrf, dpotrfbsize); } - DLLEXPORT int c_cholesky_factor(int n, cuComplex a[]){ - return cholesky_factor(n, a, cusolverDnCpotrf); + DLLEXPORT int c_cholesky_factor(cusolverDnHandle_t solverHandle, int n, cuComplex a[]) + { + return cholesky_factor(solverHandle, n, a, cpotrf, cpotrfbsize); } - DLLEXPORT int z_cholesky_factor(int n, cuDoubleComplex a[]){ - return cholesky_factor(n, a, cusolverDnZpotrf); + DLLEXPORT int z_cholesky_factor(cusolverDnHandle_t solverHandle, int n, cuDoubleComplex a[]) + { + return cholesky_factor(solverHandle, n, a, zpotrf, zpotrfbsize); } - DLLEXPORT int s_cholesky_solve(int n, int nrhs, float a[], float b[]) + DLLEXPORT int s_cholesky_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, float a[], float b[]) { - return cholesky_solve(n, nrhs, a, b, cusolverDnSpotrf, cusolverDnSpotrs); + return cholesky_solve(solverHandle, n, nrhs, a, b, spotrf, spotrs, spotrfbsize); } - DLLEXPORT int d_cholesky_solve(int n, int nrhs, double a[], double b[]) + DLLEXPORT int d_cholesky_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, double a[], double b[]) { - return cholesky_solve(n, nrhs, a, b, cusolverDnDpotrf, cusolverDnDpotrs); + return cholesky_solve(solverHandle, n, nrhs, a, b, dpotrf, dpotrs, dpotrfbsize); } - DLLEXPORT int c_cholesky_solve(int n, int nrhs, cuComplex a[], cuComplex b[]) + DLLEXPORT int c_cholesky_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, cuComplex a[], cuComplex b[]) { - return cholesky_solve(n, nrhs, a, b, cusolverDnCpotrf, cusolverDnCpotrs); + return cholesky_solve(solverHandle, n, nrhs, a, b, cpotrf, cpotrs, cpotrfbsize); } - DLLEXPORT int z_cholesky_solve(int n, int nrhs, cuDoubleComplex a[], cuDoubleComplex b[]) + DLLEXPORT int z_cholesky_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, cuDoubleComplex a[], cuDoubleComplex b[]) { - return cholesky_solve(n, nrhs, a, b, cusolverDnZpotrf, cusolverDnZpotrs); + return cholesky_solve(solverHandle, n, nrhs, a, b, zpotrf, zpotrs, zpotrfbsize); } - DLLEXPORT int s_cholesky_solve_factored(int n, int nrhs, float a[], float b[]) + DLLEXPORT int s_cholesky_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, float a[], float b[]) { - return cholesky_solve_factored(n, nrhs, a, b, cusolverDnSpotrs); + return cholesky_solve_factored(solverHandle, n, nrhs, a, b, spotrs); } - DLLEXPORT int d_cholesky_solve_factored(int n, int nrhs, double a[], double b[]) + DLLEXPORT int d_cholesky_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, double a[], double b[]) { - return cholesky_solve_factored(n, nrhs, a, b, cusolverDnDpotrs); + return cholesky_solve_factored(solverHandle, n, nrhs, a, b, dpotrs); } - DLLEXPORT int c_cholesky_solve_factored(int n, int nrhs, cuComplex a[], cuComplex b[]) + DLLEXPORT int c_cholesky_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, cuComplex a[], cuComplex b[]) { - return cholesky_solve_factored(n, nrhs, a, b, cusolverDnCpotrs); + return cholesky_solve_factored(solverHandle, n, nrhs, a, b, cpotrs); } - DLLEXPORT int z_cholesky_solve_factored(int n, int nrhs, cuDoubleComplex a[], cuDoubleComplex b[]) + DLLEXPORT int z_cholesky_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, cuDoubleComplex a[], cuDoubleComplex b[]) { - return cholesky_solve_factored(n, nrhs, a, b, cusolverDnZpotrs); + return cholesky_solve_factored(solverHandle, n, nrhs, a, b, zpotrs); } + // MJ: I am fairly certain that it would be straightforward to implement ?orgqr and ?gels but I'm focusing on getting the low-hanging fruit working first /*DLLEXPORT int s_qr_factor(int m, int n, float r[], float tau[], float q[], float work[], int len) { - return qr_factor(m, n, r, tau, q, work, len, cusolverDnSgeqrf, cusolverDnSorgqr); + return qr_factor(m, n, r, tau, q, work, len, sgeqrf, sorgqr); } DLLEXPORT int s_qr_thin_factor(int m, int n, float q[], float tau[], float r[], float work[], int len) { - return qr_thin_factor(m, n, q, tau, r, work, len, cusolverDnSgeqrf, cusolverDnSorgqr); + return qr_thin_factor(m, n, q, tau, r, work, len, sgeqrf, sorgqr); } DLLEXPORT int d_qr_factor(int m, int n, double r[], double tau[], double q[], double work[], int len) { - return qr_factor(m, n, r, tau, q, work, len, cusolverDnDgeqrf, cusolverDnDorgqr); + return qr_factor(m, n, r, tau, q, work, len, dgeqrf, dorgqr); } DLLEXPORT int d_qr_thin_factor(int m, int n, double q[], double tau[], double r[], double work[], int len) { - return qr_thin_factor(m, n, q, tau, r, work, len, cusolverDnDgeqrf, cusolverDnDorgqr); + return qr_thin_factor(m, n, q, tau, r, work, len, dgeqrf, dorgqr); } DLLEXPORT int c_qr_factor(int m, int n, cuComplex r[], cuComplex tau[], cuComplex q[], cuComplex work[], int len) { - return qr_factor(m, n, r, tau, q, work, len, cusolverDnCgeqrf, cusolverDnCungqr); + return qr_factor(m, n, r, tau, q, work, len, cgeqrf, cungqr); } DLLEXPORT int c_qr_thin_factor(int m, int n, cuComplex q[], cuComplex tau[], cuComplex r[], cuComplex work[], int len) { - return qr_thin_factor(m, n, q, tau, r, work, len, cusolverDnCgeqrf, cusolverDnCungqr); + return qr_thin_factor(m, n, q, tau, r, work, len, cgeqrf, cungqr); } - DLLEXPORT int z_qr_factor(int m, int n, cuDoubleComplex r[], cuDoubleComplex tau[], cuDoubleComplex q[]) + DLLEXPORT int z_qr_factor(int m, int n, cuDoubleComplex r[], cuDoubleComplex tau[], cuDoubleComplex q[], cuDoubleComplex work[], int len) { - return qr_factor(m, n, r, tau, q, work, len, cusolverDnZgeqrf, cusolverDnZungqr); + return qr_factor(m, n, r, tau, q, work, len, zgeqrf, zungqr); } - DLLEXPORT int z_qr_thin_factor(int m, int n, cuDoubleComplex q[], cuDoubleComplex tau[], cuDoubleComplex r[]) + DLLEXPORT int z_qr_thin_factor(int m, int n, cuDoubleComplex q[], cuDoubleComplex tau[], cuDoubleComplex r[], cuDoubleComplex work[], int len) { - return qr_thin_factor(m, n, q, tau, r, work, len, cusolverDnZgeqrf, cusolverDnZungqr); + return qr_thin_factor(m, n, q, tau, r, work, len, zgeqrf, zungqr); } DLLEXPORT int s_qr_solve(int m, int n, int bn, float a[], float b[], float x[], float work[], int len) { - return qr_solve(m, n, bn, a, b, x, work, len, sgels); + return qr_solve(m, n, bn, a, b, x, work, len, sgels); } DLLEXPORT int d_qr_solve(int m, int n, int bn, double a[], double b[], double x[], double work[], int len) { - return qr_solve(m, n, bn, a, b, x, work, len, dgels); + return qr_solve(m, n, bn, a, b, x, work, len, dgels); } DLLEXPORT int c_qr_solve(int m, int n, int bn, cuComplex a[], cuComplex b[], cuComplex x[], cuComplex work[], int len) { - return qr_solve(m, n, bn, a, b, x, work, len, cgels); + return qr_solve(m, n, bn, a, b, x, work, len, cgels); } DLLEXPORT int z_qr_solve(int m, int n, int bn, cuDoubleComplex a[], cuDoubleComplex b[], cuDoubleComplex x[], cuDoubleComplex work[], int len) { - return qr_solve(m, n, bn, a, b, x, work, len, zgels); - } + return qr_solve(m, n, bn, a, b, x, work, len, zgels); + }*/ + + //DLLEXPORT int s_qr_solve_factored(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int m, int n, int bn, float r[], float b[], float tau[], float x[], float work[], int len) + //{ + // return qr_solve_factored(solverHandle, blasHandle, m, n, bn, r, b, tau, x, work, len, sormqr, cublasStrsm); + //} + + //DLLEXPORT int d_qr_solve_factored(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int m, int n, int bn, double r[], double b[], double tau[], double x[], double work[], int len) + //{ + // return qr_solve_factored(solverHandle, blasHandle, m, n, bn, r, b, tau, x, work, len, dormqr, cublasDtrsm); + //} - DLLEXPORT int s_qr_solve_factored(int m, int n, int bn, float r[], float b[], float tau[], float x[], float work[], int len) + //DLLEXPORT int c_qr_solve_factored(int m, int n, int bn, cuComplex r[], cuComplex b[], cuComplex tau[], cuComplex x[], cuComplex work[], int len) + //{ + // return complex_qr_solve_factored(m, n, bn, r, b, tau, x, work, len, cunmqr, cublasCtrsm); + //} + + //DLLEXPORT int z_qr_solve_factored(int m, int n, int bn, cuDoubleComplex r[], cuDoubleComplex b[], cuDoubleComplex tau[], cuDoubleComplex x[], cuDoubleComplex work[], int len) + //{ + // return complex_qr_solve_factored(m, n, bn, r, b, tau, x, work, len, zunmqr, cublasZtrsm); + //} + + DLLEXPORT int s_svd_factor(cusolverDnHandle_t solverHandle, bool compute_vectors, int m, int n, float a[], float s[], float u[], float v[]) { - return qr_solve_factored(m, n, bn, r, b, tau, x, work, len, sormqr, cblas_strsm); + return svd_factor(solverHandle, compute_vectors, m, n, a, s, u, v, sgesvd, sgesvdbsize); } - DLLEXPORT int d_qr_solve_factored(int m, int n, int bn, double r[], double b[], double tau[], double x[], double work[], int len) + DLLEXPORT int d_svd_factor(cusolverDnHandle_t solverHandle, bool compute_vectors, int m, int n, double a[], double s[], double u[], double v[]) { - return qr_solve_factored(m, n, bn, r, b, tau, x, work, len, dormqr, cblas_dtrsm); + return svd_factor(solverHandle, compute_vectors, m, n, a, s, u, v,dgesvd, dgesvdbsize); } - DLLEXPORT int c_qr_solve_factored(int m, int n, int bn, cuComplex r[], cuComplex b[], cuComplex tau[], cuComplex x[], cuComplex work[], int len) + DLLEXPORT int c_svd_factor(cusolverDnHandle_t solverHandle, bool compute_vectors, int m, int n, cuComplex a[], cuComplex s[], cuComplex u[], cuComplex v[]) { - return complex_qr_solve_factored(m, n, bn, r, b, tau, x, work, len, cunmqr, cblas_ctrsm); + return complex_svd_factor(solverHandle, compute_vectors, m, n, a, s, u, v, cgesvd, cgesvdbsize); } - DLLEXPORT int z_qr_solve_factored(int m, int n, int bn, cuDoubleComplex r[], cuDoubleComplex b[], cuDoubleComplex tau[], cuDoubleComplex x[], cuDoubleComplex work[], int len) + DLLEXPORT int z_svd_factor(cusolverDnHandle_t solverHandle, bool compute_vectors, int m, int n, cuDoubleComplex a[], cuDoubleComplex s[], cuDoubleComplex u[], cuDoubleComplex v[]) { - return complex_qr_solve_factored(m, n, bn, r, b, tau, x, work, len, zunmqr, cblas_ztrsm); + return complex_svd_factor(solverHandle, compute_vectors, m, n, a, s, u, v, zgesvd, zgesvdbsize); } - DLLEXPORT int s_svd_factor(bool compute_vectors, int m, int n, float a[], float s[], float u[], float v[], float work[], int len) + /*DLLEXPORT int s_eigen(bool isSymmetric, int n, float a[], float vectors[], cuDoubleComplex values[], float d[]) { - return svd_factor(compute_vectors, m, n, a, s, u, v, work, len, sgesvd); + if (isSymmetric) + { + return sym_eigen_factor(n, a, vectors, values, d, LAPACKE_ssyev); + } + else + { + return eigen_factor(n, a, vectors, values, d, LAPACKE_sgees, LAPACKE_strevc); + } } - DLLEXPORT int d_svd_factor(bool compute_vectors, int m, int n, double a[], double s[], double u[], double v[], double work[], int len) + DLLEXPORT int d_eigen(bool isSymmetric, int n, double a[], double vectors[], cuDoubleComplex values[], double d[]) { - return svd_factor(compute_vectors, m, n, a, s, u, v, work, len, dgesvd); + if (isSymmetric) + { + return sym_eigen_factor(n, a, vectors, values, d, LAPACKE_dsyev); + } + else + { + return eigen_factor(n, a, vectors, values, d, LAPACKE_dgees, LAPACKE_dtrevc); + } } - DLLEXPORT int c_svd_factor(bool compute_vectors, int m, int n, cuComplex a[], cuComplex s[], cuComplex u[], cuComplex v[], cuComplex work[], int len) + DLLEXPORT int c_eigen(bool isSymmetric, int n, cuComplex a[], cuComplex vectors[], cuDoubleComplex values[], cuComplex d[]) { - return complex_svd_factor(compute_vectors, m, n, a, s, u, v, work, len, cgesvd); + if (isSymmetric) + { + return sym_eigen_factor(n, a, vectors, values, d, LAPACKE_cheev); + } + else + { + return eigen_complex_factor(n, a, vectors, values, d, LAPACKE_cgees, LAPACKE_ctrevc); + } } - DLLEXPORT int z_svd_factor(bool compute_vectors, int m, int n, cuDoubleComplex a[], cuDoubleComplex s[], cuDoubleComplex u[], cuDoubleComplex v[], cuDoubleComplex work[], int len) + DLLEXPORT int z_eigen(bool isSymmetric, int n, cuDoubleComplex a[], cuDoubleComplex vectors[], cuDoubleComplex values[], cuDoubleComplex d[]) { - return complex_svd_factor(compute_vectors, m, n, a, s, u, v, work, len, zgesvd); + if (isSymmetric) + { + return sym_eigen_factor(n, a, vectors, values, d, LAPACKE_zheev); + } + else + { + return eigen_complex_factor(n, a, vectors, values, d, LAPACKE_zgees, LAPACKE_ztrevc); + } }*/ -} \ No newline at end of file +} diff --git a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj index ec8944fd..a9786588 100644 --- a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj +++ b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj @@ -23,10 +23,8 @@ - - + - {5A52B796-7F41-4C90-8DE2-F3F391C4482C} @@ -118,7 +116,7 @@ true - cusolver.lib;cublas.lib;cublas_device.lib;%(AdditionalDependencies) + cudart.lib;cusolver.lib;cublas.lib;cublas_device.lib;%(AdditionalDependencies) $(CUDA_PATH)\lib\x64;%(AdditionalLibraryDirectories) @@ -152,7 +150,7 @@ true true true - cusolver.lib;cublas.lib;%(AdditionalDependencies) + cudart.lib;cusolver.lib;cublas.lib;%(AdditionalDependencies) $(CUDA_PATH)\lib\x64;%(AdditionalLibraryDirectories) diff --git a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters index 0e998fbe..9bda620d 100644 --- a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters +++ b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters @@ -23,16 +23,10 @@ Source Files - - Source Files - - - Source Files - Source Files - + Source Files From 8b7b617481103845ee7f29206c373884c435a503 Mon Sep 17 00:00:00 2001 From: Matthew Johnson Date: Sat, 25 Apr 2015 02:42:37 +0100 Subject: [PATCH 04/14] Some things are working, but most aren't. There appears to be something wrong with my usage of cublas?getriBatched, which isn't entirely surprising. I'll need to figure that out next. --- MathNet.Numerics.NativeProviders.sln | 44 +- src/NativeProviders/CUDA/blas.cpp | 6 +- src/NativeProviders/CUDA/capabilities.cpp | 79 ++ src/NativeProviders/CUDA/lapack.cpp | 125 +++- .../Windows/CUDA/CUDAWrapper.vcxproj | 18 +- .../Windows/CUDA/CUDAWrapper.vcxproj.filters | 3 + src/Numerics/Control.cs | 9 + src/Numerics/Numerics.csproj | 6 + src/Numerics/Properties/AssemblyInfo.cs | 1 + src/Numerics/Properties/Resources.Designer.cs | 21 +- src/Numerics/Properties/Resources.resx | 3 + .../Cuda/CudaLinearAlgebraProvider.Complex.cs | 703 +++++++++++++++++ .../CudaLinearAlgebraProvider.Complex32.cs | 703 +++++++++++++++++ .../Cuda/CudaLinearAlgebraProvider.Double.cs | 704 ++++++++++++++++++ .../Cuda/CudaLinearAlgebraProvider.Single.cs | 703 +++++++++++++++++ .../Cuda/CudaLinearAlgebraProvider.cs | 209 ++++++ .../LinearAlgebra/Cuda/SafeNativeMethods.cs | 378 ++++++++++ src/UnitTests/UnitTests-CUDA.csproj | 348 +++++++++ src/UnitTests/UseLinearAlgebraProvider.cs | 3 + 19 files changed, 4020 insertions(+), 46 deletions(-) create mode 100644 src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex.cs create mode 100644 src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex32.cs create mode 100644 src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Double.cs create mode 100644 src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Single.cs create mode 100644 src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs create mode 100644 src/Numerics/Providers/LinearAlgebra/Cuda/SafeNativeMethods.cs create mode 100644 src/UnitTests/UnitTests-CUDA.csproj diff --git a/MathNet.Numerics.NativeProviders.sln b/MathNet.Numerics.NativeProviders.sln index 1c7bf18d..0297e6f8 100644 --- a/MathNet.Numerics.NativeProviders.sln +++ b/MathNet.Numerics.NativeProviders.sln @@ -22,6 +22,10 @@ Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "UnitTests-MKL", "src\UnitTe EndProject Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "CUDA", "src\NativeProviders\Windows\CUDA\CUDAWrapper.vcxproj", "{5A52B796-7F41-4C90-8DE2-F3F391C4482C}" EndProject +Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "UnitTests-CUDA", "src\UnitTests\UnitTests-CUDA.csproj", "{E79C0395-01DC-4BC9-B86C-ED45790892C5}" +EndProject +Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "Scratch", "Scratch\Scratch.csproj", "{2386FAD1-BB99-4597-885C-8EF81D0637BA}" +EndProject Global GlobalSection(SolutionConfigurationPlatforms) = preSolution Debug|Any CPU = Debug|Any CPU @@ -125,13 +129,51 @@ Global {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|Mixed Platforms.Build.0 = Release|Win32 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|Win32.ActiveCfg = Release|Win32 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|Win32.Build.0 = Release|Win32 - {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|x64.ActiveCfg = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|x64.ActiveCfg = Release|x64 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|x64.Build.0 = Release|x64 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|Any CPU.ActiveCfg = Release|Win32 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|Mixed Platforms.ActiveCfg = Release|Win32 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|Mixed Platforms.Build.0 = Release|Win32 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|Win32.ActiveCfg = Release|Win32 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|Win32.Build.0 = Release|Win32 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|x64.ActiveCfg = Release|Win32 + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Debug|Any CPU.ActiveCfg = Debug|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Debug|Any CPU.Build.0 = Debug|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Debug|Mixed Platforms.ActiveCfg = Debug|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Debug|Mixed Platforms.Build.0 = Debug|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Debug|Win32.ActiveCfg = Debug|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Debug|x64.ActiveCfg = Debug|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release|Any CPU.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release|Any CPU.Build.0 = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release|Mixed Platforms.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release|Mixed Platforms.Build.0 = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release|Win32.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release|x64.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-Signed|Any CPU.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-Signed|Any CPU.Build.0 = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-Signed|Mixed Platforms.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-Signed|Mixed Platforms.Build.0 = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-Signed|Win32.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-Signed|x64.ActiveCfg = Release|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Debug|Any CPU.ActiveCfg = Debug|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Debug|Any CPU.Build.0 = Debug|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Debug|Mixed Platforms.ActiveCfg = Debug|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Debug|Mixed Platforms.Build.0 = Debug|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Debug|Win32.ActiveCfg = Debug|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Debug|x64.ActiveCfg = Debug|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Debug|x64.Build.0 = Debug|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release|Any CPU.ActiveCfg = Release|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release|Any CPU.Build.0 = Release|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release|Mixed Platforms.ActiveCfg = Release|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release|Mixed Platforms.Build.0 = Release|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release|Win32.ActiveCfg = Release|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release|x64.ActiveCfg = Release|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release-Signed|Any CPU.ActiveCfg = Release|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release-Signed|Any CPU.Build.0 = Release|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release-Signed|Mixed Platforms.ActiveCfg = Release|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release-Signed|Mixed Platforms.Build.0 = Release|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release-Signed|Win32.ActiveCfg = Release|Any CPU + {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release-Signed|x64.ActiveCfg = Release|Any CPU EndGlobalSection GlobalSection(SolutionProperties) = preSolution HideSolutionNode = FALSE diff --git a/src/NativeProviders/CUDA/blas.cpp b/src/NativeProviders/CUDA/blas.cpp index 47233b72..c8348a32 100644 --- a/src/NativeProviders/CUDA/blas.cpp +++ b/src/NativeProviders/CUDA/blas.cpp @@ -75,9 +75,8 @@ void cuda_gemm(const cublasHandle_t handle, const cublasOperation_t transa, cons cudaFree(d_C); } -#if GCC extern "C" { -#endif + DLLEXPORT void s_axpy(const cublasHandle_t blasHandle, const int n, const float alpha, const float x[], float y[]){ cuda_axpy(blasHandle, n, &alpha, x, 1, y, 1, cublasSaxpy); } @@ -162,6 +161,5 @@ extern "C" { cuda_gemm(blasHandle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m, cublasZgemm); } -#if GCC } -#endif + diff --git a/src/NativeProviders/CUDA/capabilities.cpp b/src/NativeProviders/CUDA/capabilities.cpp index e69de29b..75fb1d59 100644 --- a/src/NativeProviders/CUDA/capabilities.cpp +++ b/src/NativeProviders/CUDA/capabilities.cpp @@ -0,0 +1,79 @@ +#include "wrapper_common.h" +#include "cublas_v2.h" +#include "cusolverDn.h" + +#ifdef __cplusplus +extern "C" { +#endif /* __cplusplus */ + + /* + Capability is supported if >0 + + Actual number can be increased over time to indicate + extensions/revisions (that do not break compatibility) + */ + DLLEXPORT int query_capability(const int capability) + { + switch (capability) + { + + // SANITY CHECKS + case 0: return 0; + case 1: return -1; + + // PLATFORM + case 8: +#ifdef _M_IX86 + return 1; +#else + return 0; +#endif + case 9: +#ifdef _M_X64 + return 1; +#else + return 0; +#endif + case 10: +#ifdef _M_IA64 + return 1; +#else + return 0; +#endif + + // COMMON/SHARED + case 64: return 1; // revision + + // LINEAR ALGEBRA + case 128: return 1; // basic dense linear algebra + + // OPTIMIZATION + case 256: return 0; // basic optimization + + // FFT + case 384: return 0; // basic FFT + + default: return 0; // unknown or not supported + + } + } + + DLLEXPORT cublasStatus_t createBLASHandle(cublasHandle_t *blasHandle){ + return cublasCreate(blasHandle); + } + + DLLEXPORT cublasStatus_t destroyBLASHandle(cublasHandle_t blasHandle){ + return cublasDestroy(blasHandle); + } + + DLLEXPORT cusolverStatus_t createSolverHandle(cusolverDnHandle_t *solverHandle){ + return cusolverDnCreate(solverHandle); + } + + DLLEXPORT cusolverStatus_t destroySolverHandle(cusolverDnHandle_t solverHandle){ + return cusolverDnDestroy(solverHandle); + } + +#ifdef __cplusplus +} +#endif /* __cplusplus */ diff --git a/src/NativeProviders/CUDA/lapack.cpp b/src/NativeProviders/CUDA/lapack.cpp index 7798df12..718c4909 100644 --- a/src/NativeProviders/CUDA/lapack.cpp +++ b/src/NativeProviders/CUDA/lapack.cpp @@ -10,8 +10,6 @@ template inline int lu_factor(cusolverDnHandle_t solverHandle, int m, T a[], int ipiv[], GETRF getrf, GETRFBSIZE getrfbsize) { int info = 0; - T* work = NULL; - int lwork = 0; T* d_A = NULL; cudaMalloc((void**)&d_A, m*m*sizeof(T)); @@ -20,10 +18,17 @@ inline int lu_factor(cusolverDnHandle_t solverHandle, int m, T a[], int ipiv[], int* d_I = NULL; cudaMalloc((void**)&d_I, m*sizeof(int)); + T* work = NULL; + int lwork = 0; getrfbsize(solverHandle, m, m, a, m, &lwork); - cudaMalloc((void**)lwork, sizeof(T)*lwork); + cudaMalloc((void**)&work, sizeof(T)*lwork); + + int* d_info = NULL; + cudaMalloc((void**)&d_info, sizeof(int)); - getrf(solverHandle, m, m, d_A, m, work, d_I, &info); + getrf(solverHandle, m, m, d_A, m, work, d_I, d_info); + + cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); cublasGetMatrix(m, m, sizeof(T), d_A, m, a, m); cublasGetVector(m, sizeof(T), d_I, 1, ipiv, 1); @@ -32,6 +37,7 @@ inline int lu_factor(cusolverDnHandle_t solverHandle, int m, T a[], int ipiv[], cudaFree(d_A); cudaFree(d_I); + cudaFree(d_info); cudaFree(work); return info; @@ -41,8 +47,6 @@ template inline int lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int n, T a[], GETRF getrf, GETRI getri, GETRFBSIZE getrfbsize) { int info = 0; - T* work = NULL; - int lwork = 0; int* d_I = NULL; cudaMalloc((void**)&d_I, n*sizeof(T)); @@ -51,30 +55,48 @@ inline int lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle cudaMalloc((void**)&d_A, n*n*sizeof(T)); cublasSetMatrix(n, n, sizeof(T), a, n, d_A, n); + T* work = NULL; + int lwork = 0; getrfbsize(solverHandle, n, n, d_A, n, &lwork); - cudaMalloc((void**)lwork, sizeof(T)*lwork); + cudaMalloc((void**)&work, sizeof(T)*lwork); + + int* d_info = NULL; + cudaMalloc((void**)&d_info, sizeof(int)); + + printf("initial %f %f %f %f %f %f %f %f %f\r\n", a[0], a[1], a[2], a[3], a[4], a[5], a[6], a[7], a[8]); - getrf(solverHandle, n, n, d_A, n, work, d_I, &info); + getrf(solverHandle, n, n, d_A, n, work, d_I, d_info); + cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); + cublasGetMatrix(n, n, sizeof(T), d_A, n, a, n); + printf("after factor %f %f %f %f %f %f %f %f %f\r\n", a[0], a[1], a[2], a[3], a[4], a[5], a[6], a[7], a[8]); + cudaFree(work); if (info != 0) { cudaFree(d_A); cudaFree(d_I); + cudaFree(d_info); return info; } T* d_C = NULL; cudaMalloc((void**)&d_C, n*n*sizeof(T)); - getri(blasHandle, n, d_A, n, d_I, d_C, n, &info); + getri(blasHandle, n, d_A, n, d_I, d_C, n, d_info); + cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); + + cublasGetMatrix(n, n, sizeof(T), d_A, n, a, n); + printf("a inverse %f %f %f %f %f %f %f %f %f\r\n", a[0], a[1], a[2], a[3], a[4], a[5], a[6], a[7], a[8]); cublasGetMatrix(n, n, sizeof(T), d_C, n, a, n); + printf("c inverse %f %f %f %f %f %f %f %f %f\r\n", a[0], a[1], a[2], a[3], a[4], a[5], a[6], a[7], a[8]); cudaFree(d_A); cudaFree(d_I); cudaFree(d_C); + cudaFree(d_info); return info; }; @@ -82,9 +104,10 @@ inline int lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle template inline int lu_inverse_factored(cublasHandle_t blasHandle, int n, T a[], int ipiv[], GETRI getri) { - shift_ipiv_up(n, ipiv); int info = 0; + shift_ipiv_up(n, ipiv); + T* d_A = NULL; cudaMalloc((void**)&d_A, n*n*sizeof(T)); cublasSetMatrix(n, n, sizeof(T), a, n, d_A, n); @@ -96,7 +119,11 @@ inline int lu_inverse_factored(cublasHandle_t blasHandle, int n, T a[], int ipiv cudaMalloc((void**)&d_I, n*sizeof(int)); cublasSetVector(n, sizeof(int), ipiv, 1, d_I, 1); - getri(blasHandle, n, d_A, n, d_I, d_C, n, &info); + int* d_info = NULL; + cudaMalloc((void**)&d_info, sizeof(int)); + + getri(blasHandle, n, d_A, n, d_I, d_C, n, d_info); + cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); cublasGetMatrix(n, n, sizeof(T), d_C, n, a, n); cublasGetVector(n, sizeof(int), d_I, 1, ipiv, 1); @@ -106,6 +133,7 @@ inline int lu_inverse_factored(cublasHandle_t blasHandle, int n, T a[], int ipiv cudaFree(d_A); cudaFree(d_I); cudaFree(d_C); + cudaFree(d_info); return info; } @@ -113,9 +141,10 @@ inline int lu_inverse_factored(cublasHandle_t blasHandle, int n, T a[], int ipiv template inline int lu_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[], int ipiv[], T b[], GETRS getrs) { - shift_ipiv_up(n, ipiv); int info = 0; + shift_ipiv_up(n, ipiv); + T* d_A = NULL; cudaMalloc((void**)&d_A, n*n*sizeof(T)); cublasSetMatrix(n, n, sizeof(T), a, n, d_A, n); @@ -128,7 +157,11 @@ inline int lu_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, T cudaMalloc((void**)&d_I, n*sizeof(int)); cublasSetVector(n, sizeof(int), ipiv, 1, d_I, 1); - getrs(solverHandle, CUBLAS_OP_N, n, nrhs, d_A, n, d_I, d_B, n, &info); + int* d_info = NULL; + cudaMalloc((void**)&d_info, sizeof(int)); + + getrs(solverHandle, CUBLAS_OP_N, n, nrhs, d_A, n, d_I, d_B, n, d_info); + cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); cublasGetMatrix(n, nrhs, sizeof(T), d_B, n, b, n); @@ -137,6 +170,7 @@ inline int lu_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, T cudaFree(d_A); cudaFree(d_B); cudaFree(d_I); + cudaFree(d_info); return info; } @@ -145,8 +179,6 @@ template inline int lu_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[], T b[], GETRF getrf, GETRS getrs, GETRFBSIZE getrfbsize) { int info = 0; - T* work = NULL; - int lwork = 0; int* d_I = NULL; cudaMalloc((void**)&d_I, n*sizeof(T)); @@ -155,15 +187,22 @@ inline int lu_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[], T b cudaMalloc((void**)&d_A, n*n*sizeof(T)); cublasSetMatrix(n, n, sizeof(T), a, n, d_A, n); + T* work = NULL; + int lwork = 0; getrfbsize(solverHandle, n, n, a, n, &lwork); - cudaMalloc((void**)lwork, sizeof(T)*lwork); + cudaMalloc((void**)&work, sizeof(T)*lwork); + + int* d_info = NULL; + cudaMalloc((void**)&d_info, sizeof(int)); - getrf(solverHandle, n, n, d_A, n, work, d_I, &info); + getrf(solverHandle, n, n, d_A, n, work, d_I, d_info); + cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); if (info != 0) { cudaFree(d_I); cudaFree(d_A); + cudaFree(d_info); return info; } @@ -171,13 +210,15 @@ inline int lu_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[], T b cudaMalloc((void**)&d_B, n*nrhs*sizeof(T)); cublasSetMatrix(n, nrhs, sizeof(T), b, n, d_B, n); - getrs(solverHandle, CUBLAS_OP_N, n, nrhs, d_A, n, d_I, d_B, n, &info); + getrs(solverHandle, CUBLAS_OP_N, n, nrhs, d_A, n, d_I, d_B, n, d_info); + cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); cublasGetMatrix(n, nrhs, sizeof(T), d_B, n, b, n); cudaFree(d_A); cudaFree(d_B); cudaFree(d_I); + cudaFree(d_info); return info; } @@ -197,7 +238,11 @@ inline int cholesky_factor(cusolverDnHandle_t solverHandle, int n, T* a, POTRF p potrfbsize(solverHandle, CUBLAS_FILL_MODE_LOWER, n, d_A, n, &lWork); cudaMalloc((void**)&work, sizeof(T)*lWork); - potrf(solverHandle, CUBLAS_FILL_MODE_LOWER, n, d_A, n, work, lWork, &info); + int* d_info = NULL; + cudaMalloc((void**)&d_info, sizeof(int)); + + potrf(solverHandle, CUBLAS_FILL_MODE_LOWER, n, d_A, n, work, lWork, d_info); + cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); cublasGetMatrix(n, n, sizeof(T), d_A, n, a, n); @@ -214,6 +259,7 @@ inline int cholesky_factor(cusolverDnHandle_t solverHandle, int n, T* a, POTRF p } cudaFree(d_A); + cudaFree(d_info); cudaFree(work); return info; @@ -233,13 +279,18 @@ inline int cholesky_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[ potrfbsize(solverHandle, CUBLAS_FILL_MODE_LOWER, n, d_A, n, &lWork); cudaMalloc((void**)&work, sizeof(T)*lWork); - potrf(solverHandle, CUBLAS_FILL_MODE_LOWER, n, d_A, n, work, lWork, &info); + int* d_info = NULL; + cudaMalloc((void**)&d_info, sizeof(int)); + + potrf(solverHandle, CUBLAS_FILL_MODE_LOWER, n, d_A, n, work, lWork, d_info); + cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); cudaFree(work); if (info != 0) { cudaFree(d_A); + cudaFree(d_info); return info; } @@ -247,12 +298,14 @@ inline int cholesky_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[ cudaMalloc((void**)d_B, n*nrhs*sizeof(T)); cublasSetMatrix(n, nrhs, sizeof(T), b, n, d_B, n); - potrs(solverHandle, CUBLAS_FILL_MODE_LOWER, n, nrhs, d_A, n, d_B, n, &info); + potrs(solverHandle, CUBLAS_FILL_MODE_LOWER, n, nrhs, d_A, n, d_B, n, d_info); + cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); cublasGetMatrix(n, nrhs, sizeof(T), d_B, n, b, n); cudaFree(d_A); cudaFree(d_B); + cudaFree(d_info); return info; } @@ -270,12 +323,17 @@ inline int cholesky_solve_factored(cusolverDnHandle_t solverHandle, int n, int n cudaMalloc((void**)d_B, n*nrhs*sizeof(T)); cublasSetMatrix(n, nrhs, sizeof(T), b, n, d_B, n); - potrs(solverHandle, CUBLAS_FILL_MODE_LOWER, n, nrhs, d_A, n, d_B, n, &info); + int* d_info = NULL; + cudaMalloc((void**)&d_info, sizeof(int)); + + potrs(solverHandle, CUBLAS_FILL_MODE_LOWER, n, nrhs, d_A, n, d_B, n, d_info); + cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); cublasGetMatrix(n, nrhs, sizeof(T), d_B, n, b, n); cudaFree(d_A); cudaFree(d_B); + cudaFree(d_info); return info; } @@ -402,8 +460,13 @@ inline int svd_factor(cusolverDnHandle_t solverHandle, bool compute_vectors, int T* rwork = NULL; cudaMalloc((void**)&rwork, 5 * dim_s * sizeof(T)); + int* d_info = NULL; + cudaMalloc((void**)&d_info, sizeof(int)); + + char job = compute_vectors ? 'A' : 'N'; - gesvd(solverHandle, job, job, m, n, d_A, m, d_S, d_U, m, d_V, n, work, lWork, rwork, &info); + gesvd(solverHandle, job, job, m, n, d_A, m, d_S, d_U, m, d_V, n, work, lWork, rwork, d_info); + cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); cublasGetVector(dim_s, sizeof(T), d_S, 1, s, 1); cublasGetMatrix(m, m, sizeof(T), d_U, m, u, m); @@ -415,6 +478,7 @@ inline int svd_factor(cusolverDnHandle_t solverHandle, bool compute_vectors, int cudaFree(d_V); cudaFree(work); cudaFree(rwork); + cudaFree(d_info); return info; } @@ -447,8 +511,12 @@ inline int complex_svd_factor(cusolverDnHandle_t solverHandle, bool compute_vect R* rwork = NULL; cudaMalloc((void**)&rwork, 5 * dim_s * sizeof(R)); + int* d_info = NULL; + cudaMalloc((void**)&d_info, sizeof(int)); + char job = compute_vectors ? 'A' : 'N'; - gesvd(solverHandle, job, job, m, n, d_A, m, d_S, d_U, m, d_V, n, work, lWork, rwork, &info); + gesvd(solverHandle, job, job, m, n, d_A, m, d_S, d_U, m, d_V, n, work, lWork, rwork, d_info); + cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); cublasGetVector(dim_s, sizeof(T), d_S, 1, s_local, 1); cublasGetMatrix(m, m, sizeof(T), d_U, m, u, m); @@ -466,6 +534,7 @@ inline int complex_svd_factor(cusolverDnHandle_t solverHandle, bool compute_vect cudaFree(d_V); cudaFree(work); cudaFree(rwork); + cudaFree(d_info); return info; } @@ -643,22 +712,22 @@ inline int complex_svd_factor(cusolverDnHandle_t solverHandle, bool compute_vect #define zgesvdbsize cusolverDnZgesvd_bufferSize -inline int sgetri(cublasHandle_t handle, int n, const float a[], int lda, const int *ipiv, float c[], int ldc, int *info) +inline int sgetri(cublasHandle_t handle, int n, const float a[], int lda, const int ipiv[], float c[], int ldc, int *info) { return cublasSgetriBatched(handle, n, &a, lda, ipiv, &c, ldc, info, 1); } -inline int dgetri(cublasHandle_t handle, int n, const double a[], int lda, const int *ipiv, double c[], int ldc, int *info) +inline int dgetri(cublasHandle_t handle, int n, const double a[], int lda, const int ipiv[], double c[], int ldc, int *info) { return cublasDgetriBatched(handle, n, &a, lda, ipiv, &c, ldc, info, 1); } -inline int cgetri(cublasHandle_t handle, int n, const cuComplex a[], int lda, const int *ipiv, cuComplex c[], int ldc, int *info) +inline int cgetri(cublasHandle_t handle, int n, const cuComplex a[], int lda, const int ipiv[], cuComplex c[], int ldc, int *info) { return cublasCgetriBatched(handle, n, &a, lda, ipiv, &c, ldc, info, 1); } -inline int zgetri(cublasHandle_t handle, int n, const cuDoubleComplex a[], int lda, const int *ipiv, cuDoubleComplex c[], int ldc, int *info) +inline int zgetri(cublasHandle_t handle, int n, const cuDoubleComplex a[], int lda, const int ipiv[], cuDoubleComplex c[], int ldc, int *info) { return cublasZgetriBatched(handle, n, &a, lda, ipiv, &c, ldc, info, 1); } diff --git a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj index a9786588..393b8c88 100644 --- a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj +++ b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj @@ -24,6 +24,7 @@ + @@ -111,14 +112,20 @@ Level3 Disabled - true $(CUDA_PATH)\include;$(ProjectDir)..\..\Common;$(ProjectDir)..\..\CUDA;%(AdditionalIncludeDirectories) + _WINDOWS;%(PreprocessorDefinitions) + MultiThreadedDebug true cudart.lib;cusolver.lib;cublas.lib;cublas_device.lib;%(AdditionalDependencies) $(CUDA_PATH)\lib\x64;%(AdditionalLibraryDirectories) + + copy "$(CUDA_PATH)\bin\cublas64_70.dll" $(OutputPath) +copy "$(CUDA_PATH)\bin\cusolver64_70.dll" $(OutputPath) +copy "$(CUDA_PATH)\bin\cudart64_70.dll" $(OutputPath) + @@ -143,8 +150,10 @@ MaxSpeed true true - true $(CUDA_PATH)\include;$(ProjectDir)..\..\Common;$(ProjectDir)..\..\CUDA;%(AdditionalIncludeDirectories) + _WINDOWS;%(PreprocessorDefinitions) + /Qvec-report:1 + MultiThreaded true @@ -153,6 +162,11 @@ cudart.lib;cusolver.lib;cublas.lib;%(AdditionalDependencies) $(CUDA_PATH)\lib\x64;%(AdditionalLibraryDirectories) + + copy "$(CUDA_PATH)\bin\cublas64_70.dll" $(OutputPath) +copy "$(CUDA_PATH)\bin\cusolver64_70.dll" $(OutputPath) +copy "$(CUDA_PATH)\bin\cudart64_70.dll" $(OutputPath) + diff --git a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters index 9bda620d..3c9be8c5 100644 --- a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters +++ b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters @@ -29,5 +29,8 @@ Source Files + + Source Files + \ No newline at end of file diff --git a/src/Numerics/Control.cs b/src/Numerics/Control.cs index fa1028fc..0e9570fd 100644 --- a/src/Numerics/Control.cs +++ b/src/Numerics/Control.cs @@ -78,6 +78,10 @@ namespace MathNet.Numerics case "MKL": LinearAlgebraProvider = new Providers.LinearAlgebra.Mkl.MklLinearAlgebraProvider(); break; + + case "CUDA": + LinearAlgebraProvider = new Providers.LinearAlgebra.Cuda.CudaLinearAlgebraProvider(); + break; #endif default: LinearAlgebraProvider = new ManagedLinearAlgebraProvider(); @@ -127,6 +131,11 @@ namespace MathNet.Numerics { LinearAlgebraProvider = new Providers.LinearAlgebra.Mkl.MklLinearAlgebraProvider(consistency, precision, accuracy); } + + public static void UseNativeCUDA() + { + LinearAlgebraProvider = new Providers.LinearAlgebra.Cuda.CudaLinearAlgebraProvider(); + } #endif /// diff --git a/src/Numerics/Numerics.csproj b/src/Numerics/Numerics.csproj index b783234e..b451cf78 100644 --- a/src/Numerics/Numerics.csproj +++ b/src/Numerics/Numerics.csproj @@ -157,6 +157,12 @@ + + + + + + diff --git a/src/Numerics/Properties/AssemblyInfo.cs b/src/Numerics/Properties/AssemblyInfo.cs index 2281025c..930f5179 100644 --- a/src/Numerics/Properties/AssemblyInfo.cs +++ b/src/Numerics/Properties/AssemblyInfo.cs @@ -76,6 +76,7 @@ using System.Runtime.InteropServices; #else [assembly: InternalsVisibleTo("MathNet.Numerics.UnitTests")] [assembly: InternalsVisibleTo("MathNet.Numerics.UnitTestsMKL")] +[assembly: InternalsVisibleTo("MathNet.Numerics.UnitTestsCUDA")] [assembly: InternalsVisibleTo("Performance")] #endif diff --git a/src/Numerics/Properties/Resources.Designer.cs b/src/Numerics/Properties/Resources.Designer.cs index 0c7fae4d..443ba150 100644 --- a/src/Numerics/Properties/Resources.Designer.cs +++ b/src/Numerics/Properties/Resources.Designer.cs @@ -1,15 +1,13 @@ //------------------------------------------------------------------------------ // // This code was generated by a tool. -// Runtime Version:4.0.30319.34209 +// Runtime Version:4.0.30319.34014 // // Changes to this file may cause incorrect behavior and will be lost if // the code is regenerated. // //------------------------------------------------------------------------------ -using System.Reflection; - namespace MathNet.Numerics.Properties { using System; @@ -40,18 +38,10 @@ namespace MathNet.Numerics.Properties { [global::System.ComponentModel.EditorBrowsableAttribute(global::System.ComponentModel.EditorBrowsableState.Advanced)] public static global::System.Resources.ResourceManager ResourceManager { get { -#if NET45REFLECTION - if (object.ReferenceEquals(resourceMan, null)) - { - global::System.Resources.ResourceManager temp = new global::System.Resources.ResourceManager("MathNet.Numerics.Properties.Resources", typeof(Resources).GetTypeInfo().Assembly); - resourceMan = temp; - } -#else if (object.ReferenceEquals(resourceMan, null)) { global::System.Resources.ResourceManager temp = new global::System.Resources.ResourceManager("MathNet.Numerics.Properties.Resources", typeof(Resources).Assembly); resourceMan = temp; } -#endif return resourceMan; } } @@ -935,6 +925,15 @@ namespace MathNet.Numerics.Properties { } } + /// + /// Looks up a localized string similar to User work buffers are not supported by this provider.. + /// + public static string UserWorkBufferNotSupported { + get { + return ResourceManager.GetString("UserWorkBufferNotSupported", resourceCulture); + } + } + /// /// Looks up a localized string similar to Vectors can not be empty and must have at least one element.. /// diff --git a/src/Numerics/Properties/Resources.resx b/src/Numerics/Properties/Resources.resx index d46d44c8..e1360503 100644 --- a/src/Numerics/Properties/Resources.resx +++ b/src/Numerics/Properties/Resources.resx @@ -412,4 +412,7 @@ Vectors can not be empty and must have at least one element. + + User work buffers are not supported by this provider. + \ No newline at end of file diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex.cs new file mode 100644 index 00000000..bad3b091 --- /dev/null +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex.cs @@ -0,0 +1,703 @@ +// +// Math.NET Numerics, part of the Math.NET Project +// http://numerics.mathdotnet.com +// http://github.com/mathnet/mathnet-numerics +// http://mathnetnumerics.codeplex.com +// +// Copyright (c) 2009-2013 Math.NET +// +// Permission is hereby granted, free of charge, to any person +// obtaining a copy of this software and associated documentation +// files (the "Software"), to deal in the Software without +// restriction, including without limitation the rights to use, +// copy, modify, merge, publish, distribute, sublicense, and/or sell +// copies of the Software, and to permit persons to whom the +// Software is furnished to do so, subject to the following +// conditions: +// +// The above copyright notice and this permission notice shall be +// included in all copies or substantial portions of the Software. +// +// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, +// EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES +// OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND +// NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT +// HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, +// WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +// FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR +// OTHER DEALINGS IN THE SOFTWARE. +// + +#if NATIVE + +using System; +using System.Numerics; +using System.Security; +using MathNet.Numerics.LinearAlgebra.Factorization; +using MathNet.Numerics.Properties; + +namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda +{ + /// + /// Intel's Math Kernel Library (MKL) linear algebra provider. + /// + public partial class CudaLinearAlgebraProvider + { + /// + /// Computes the dot product of x and y. + /// + /// The vector x. + /// The vector y. + /// The dot product of x and y. + /// This is equivalent to the DOT BLAS routine. + [SecuritySafeCritical] + public override Complex DotProduct(Complex[] x, Complex[] y) + { + if (y == null) + { + throw new ArgumentNullException("y"); + } + + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (x.Length != y.Length) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength); + } + + return SafeNativeMethods.z_dot_product(_blasHandle, x.Length, x, y); + } + + /// + /// Adds a scaled vector to another: result = y + alpha*x. + /// + /// The vector to update. + /// The value to scale by. + /// The vector to add to . + /// The result of the addition. + /// This is similar to the AXPY BLAS routine. + [SecuritySafeCritical] + public override void AddVectorToScaledVector(Complex[] y, Complex alpha, Complex[] x, Complex[] result) + { + if (y == null) + { + throw new ArgumentNullException("y"); + } + + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (y.Length != x.Length) + { + throw new ArgumentException(Resources.ArgumentVectorsSameLength); + } + + if (!ReferenceEquals(y, result)) + { + Array.Copy(y, 0, result, 0, y.Length); + } + + if (alpha == Complex.Zero) + { + return; + } + + SafeNativeMethods.z_axpy(_blasHandle, y.Length, alpha, x, result); + } + + /// + /// Scales an array. Can be used to scale a vector and a matrix. + /// + /// The scalar. + /// The values to scale. + /// This result of the scaling. + /// This is similar to the SCAL BLAS routine. + [SecuritySafeCritical] + public override void ScaleArray(Complex alpha, Complex[] x, Complex[] result) + { + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (!ReferenceEquals(x, result)) + { + Array.Copy(x, 0, result, 0, x.Length); + } + + if (alpha == Complex.One) + { + return; + } + + SafeNativeMethods.z_scale(_blasHandle, x.Length, alpha, result); + } + + /// + /// Multiples two matrices. result = x * y + /// + /// The x matrix. + /// The number of rows in the x matrix. + /// The number of columns in the x matrix. + /// The y matrix. + /// The number of rows in the y matrix. + /// The number of columns in the y matrix. + /// Where to store the result of the multiplication. + /// This is a simplified version of the BLAS GEMM routine with alpha + /// set to Complex.One and beta set to Complex.Zero, and x and y are not transposed. + public override void MatrixMultiply(Complex[] x, int rowsX, int columnsX, Complex[] y, int rowsY, int columnsY, Complex[] result) + { + MatrixMultiplyWithUpdate(Transpose.DontTranspose, Transpose.DontTranspose, Complex.One, x, rowsX, columnsX, y, rowsY, columnsY, Complex.Zero, result); + } + + /// + /// Multiplies two matrices and updates another with the result. c = alpha*op(a)*op(b) + beta*c + /// + /// How to transpose the matrix. + /// How to transpose the matrix. + /// The value to scale matrix. + /// The a matrix. + /// The number of rows in the matrix. + /// The number of columns in the matrix. + /// The b matrix + /// The number of rows in the matrix. + /// The number of columns in the matrix. + /// The value to scale the matrix. + /// The c matrix. + [SecuritySafeCritical] + public override void MatrixMultiplyWithUpdate(Transpose transposeA, Transpose transposeB, Complex alpha, Complex[] a, int rowsA, int columnsA, Complex[] b, int rowsB, int columnsB, Complex beta, Complex[] c) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (c == null) + { + throw new ArgumentNullException("c"); + } + + var m = transposeA == Transpose.DontTranspose ? rowsA : columnsA; + var n = transposeB == Transpose.DontTranspose ? columnsB : rowsB; + var k = transposeA == Transpose.DontTranspose ? columnsA : rowsA; + var l = transposeB == Transpose.DontTranspose ? rowsB : columnsB; + + if (c.Length != m*n) + { + throw new ArgumentException(Resources.ArgumentMatrixDimensions); + } + + if (k != l) + { + throw new ArgumentException(Resources.ArgumentMatrixDimensions); + } + + SafeNativeMethods.z_matrix_multiply(_blasHandle, transposeA.ToCUDA(), transposeB.ToCUDA(), m, n, k, alpha, a, b, beta, c); + } + + /// + /// Computes the LUP factorization of A. P*A = L*U. + /// + /// An by matrix. The matrix is overwritten with the + /// the LU factorization on exit. The lower triangular factor L is stored in under the diagonal of (the diagonal is always Complex.One + /// for the L factor). The upper triangular factor U is stored on and above the diagonal of . + /// The order of the square matrix . + /// On exit, it contains the pivot indices. The size of the array must be . + /// This is equivalent to the GETRF LAPACK routine. + [SecuritySafeCritical] + public override void LUFactor(Complex[] data, int order, int[] ipiv) + { + if (data == null) + { + throw new ArgumentNullException("data"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (data.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "data"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + Solver(SafeNativeMethods.z_lu_factor(_solverHandle, order, data, ipiv)); + } + + /// + /// Computes the inverse of matrix using LU factorization. + /// + /// The N by N matrix to invert. Contains the inverse On exit. + /// The order of the square matrix . + /// This is equivalent to the GETRF and GETRI LAPACK routines. + [SecuritySafeCritical] + public override void LUInverse(Complex[] a, int order) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + Solver(SafeNativeMethods.z_lu_inverse(_solverHandle, _blasHandle, order, a)); + } + + /// + /// Computes the inverse of a previously factored matrix. + /// + /// The LU factored N by N matrix. Contains the inverse On exit. + /// The order of the square matrix . + /// The pivot indices of . + /// This is equivalent to the GETRI LAPACK routine. + [SecuritySafeCritical] + public override void LUInverseFactored(Complex[] a, int order, int[] ipiv) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + BLAS(SafeNativeMethods.z_lu_inverse_factored(_blasHandle, order, a, ipiv)); + } + + /// + /// Computes the inverse of matrix using LU factorization. + /// + /// The N by N matrix to invert. Contains the inverse On exit. + /// The order of the square matrix . + /// Not supported. Should be left null. + /// This is equivalent to the GETRF and GETRI LAPACK routines. + [SecuritySafeCritical] + public override void LUInverse(Complex[] a, int order, Complex[] work) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (work != null) + { + throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + } + + Solver(SafeNativeMethods.z_lu_inverse(_solverHandle, _blasHandle, order, a)); + } + + /// + /// Computes the inverse of a previously factored matrix. + /// + /// The LU factored N by N matrix. Contains the inverse On exit. + /// The order of the square matrix . + /// The pivot indices of . + /// Not supported. Should be left null. + /// This is equivalent to the GETRI LAPACK routine. + [SecuritySafeCritical] + public override void LUInverseFactored(Complex[] a, int order, int[] ipiv, Complex[] work) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + if (work != null) + { + throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + } + + BLAS(SafeNativeMethods.z_lu_inverse_factored(_blasHandle, order, a, ipiv)); + } + + /// + /// Solves A*X=B for X using LU factorization. + /// + /// The number of columns of B. + /// The square matrix A. + /// The order of the square matrix . + /// On entry the B matrix; on exit the X matrix. + /// This is equivalent to the GETRF and GETRS LAPACK routines. + [SecuritySafeCritical] + public override void LUSolve(int columnsOfB, Complex[] a, int order, Complex[] b) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (b.Length != columnsOfB*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.z_lu_solve(_solverHandle, order, columnsOfB, a, b)); + } + + /// + /// Solves A*X=B for X using a previously factored A matrix. + /// + /// The number of columns of B. + /// The factored A matrix. + /// The order of the square matrix . + /// The pivot indices of . + /// On entry the B matrix; on exit the X matrix. + /// This is equivalent to the GETRS LAPACK routine. + [SecuritySafeCritical] + public override void LUSolveFactored(int columnsOfB, Complex[] a, int order, int[] ipiv, Complex[] b) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + if (b.Length != columnsOfB*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.z_lu_solve_factored(_solverHandle, order, columnsOfB, a, ipiv, b)); + } + + /// + /// Computes the Cholesky factorization of A. + /// + /// On entry, a square, positive definite matrix. On exit, the matrix is overwritten with the + /// the Cholesky factorization. + /// The number of rows or columns in the matrix. + /// This is equivalent to the POTRF LAPACK routine. + [SecuritySafeCritical] + public override void CholeskyFactor(Complex[] a, int order) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (order < 1) + { + throw new ArgumentException(Resources.ArgumentMustBePositive, "order"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + Solver(SafeNativeMethods.z_cholesky_factor(_solverHandle, order, a)); + } + + /// + /// Solves A*X=B for X using Cholesky factorization. + /// + /// The square, positive definite matrix A. + /// The number of rows and columns in A. + /// On entry the B matrix; on exit the X matrix. + /// The number of columns in the B matrix. + /// This is equivalent to the POTRF add POTRS LAPACK routines. + /// + [SecuritySafeCritical] + public override void CholeskySolve(Complex[] a, int orderA, Complex[] b, int columnsB) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (b.Length != orderA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.z_cholesky_solve(_solverHandle, orderA, columnsB, a, b)); + } + + /// + /// Solves A*X=B for X using a previously factored A matrix. + /// + /// The square, positive definite matrix A. + /// The number of rows and columns in A. + /// On entry the B matrix; on exit the X matrix. + /// The number of columns in the B matrix. + /// This is equivalent to the POTRS LAPACK routine. + [SecuritySafeCritical] + public override void CholeskySolveFactored(Complex[] a, int orderA, Complex[] b, int columnsB) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (b.Length != orderA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.z_cholesky_solve_factored(_solverHandle, orderA, columnsB, a, b)); + } + + /// + /// Computes the singular value decomposition of A. + /// + /// Compute the singular U and VT vectors or not. + /// On entry, the M by N matrix to decompose. On exit, A may be overwritten. + /// The number of rows in the A matrix. + /// The number of columns in the A matrix. + /// The singular values of A in ascending value. + /// If is true, on exit U contains the left + /// singular vectors. + /// If is true, on exit VT contains the transposed + /// right singular vectors. + /// This is equivalent to the GESVD LAPACK routine. + [SecuritySafeCritical] + public override void SingularValueDecomposition(bool computeVectors, Complex[] a, int rowsA, int columnsA, Complex[] s, Complex[] u, Complex[] vt) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (s == null) + { + throw new ArgumentNullException("s"); + } + + if (u == null) + { + throw new ArgumentNullException("u"); + } + + if (vt == null) + { + throw new ArgumentNullException("vt"); + } + + if (u.Length != rowsA*rowsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "u"); + } + + if (vt.Length != columnsA*columnsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "vt"); + } + + if (s.Length != Math.Min(rowsA, columnsA)) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "s"); + } + + SingularValueDecomposition(computeVectors, a, rowsA, columnsA, s, u, vt, null); + } + + /// + /// Solves A*X=B for X using the singular value decomposition of A. + /// + /// On entry, the M by N matrix to decompose. + /// The number of rows in the A matrix. + /// The number of columns in the A matrix. + /// The B matrix. + /// The number of columns of B. + /// On exit, the solution matrix. + public override void SvdSolve(Complex[] a, int rowsA, int columnsA, Complex[] b, int columnsB, Complex[] x) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (b.Length != rowsA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (x.Length != columnsA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + var s = new Complex[Math.Min(rowsA, columnsA)]; + var u = new Complex[rowsA*rowsA]; + var vt = new Complex[columnsA*columnsA]; + + var clone = new Complex[a.Length]; + a.Copy(clone); + SingularValueDecomposition(true, clone, rowsA, columnsA, s, u, vt, null); + SvdSolveFactored(rowsA, columnsA, s, u, vt, b, columnsB, x); + } + + /// + /// Computes the singular value decomposition of A. + /// + /// Compute the singular U and VT vectors or not. + /// On entry, the M by N matrix to decompose. On exit, A may be overwritten. + /// The number of rows in the A matrix. + /// The number of columns in the A matrix. + /// The singular values of A in ascending value. + /// If is true, on exit U contains the left + /// singular vectors. + /// If is true, on exit VT contains the transposed + /// right singular vectors. + /// User work buffers are not supported. Should be null. + /// This is equivalent to the GESVD LAPACK routine. + [SecuritySafeCritical] + public override void SingularValueDecomposition(bool computeVectors, Complex[] a, int rowsA, int columnsA, Complex[] s, Complex[] u, Complex[] vt, Complex[] work) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (s == null) + { + throw new ArgumentNullException("s"); + } + + if (u == null) + { + throw new ArgumentNullException("u"); + } + + if (vt == null) + { + throw new ArgumentNullException("vt"); + } + + if (work != null) + { + throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + } + + if (u.Length != rowsA*rowsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "u"); + } + + if (vt.Length != columnsA*columnsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "vt"); + } + + if (s.Length != Math.Min(rowsA, columnsA)) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "s"); + } + + Solver(SafeNativeMethods.z_svd_factor(_solverHandle, computeVectors, rowsA, columnsA, a, s, u, vt)); + } + } +} + +#endif diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex32.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex32.cs new file mode 100644 index 00000000..3e06022c --- /dev/null +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex32.cs @@ -0,0 +1,703 @@ +// +// Math.NET Numerics, part of the Math.NET Project +// http://numerics.mathdotnet.com +// http://github.com/mathnet/mathnet-numerics +// http://mathnetnumerics.codeplex.com +// +// Copyright (c) 2009-2013 Math.NET +// +// Permission is hereby granted, free of charge, to any person +// obtaining a copy of this software and associated documentation +// files (the "Software"), to deal in the Software without +// restriction, including without limitation the rights to use, +// copy, modify, merge, publish, distribute, sublicense, and/or sell +// copies of the Software, and to permit persons to whom the +// Software is furnished to do so, subject to the following +// conditions: +// +// The above copyright notice and this permission notice shall be +// included in all copies or substantial portions of the Software. +// +// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, +// EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES +// OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND +// NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT +// HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, +// WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +// FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR +// OTHER DEALINGS IN THE SOFTWARE. +// + +#if NATIVE + +using System; +using System.Numerics; +using System.Security; +using MathNet.Numerics.LinearAlgebra.Factorization; +using MathNet.Numerics.Properties; + +namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda +{ + /// + /// Intel's Math Kernel Library (MKL) linear algebra provider. + /// + public partial class CudaLinearAlgebraProvider + { + /// + /// Computes the dot product of x and y. + /// + /// The vector x. + /// The vector y. + /// The dot product of x and y. + /// This is equivalent to the DOT BLAS routine. + [SecuritySafeCritical] + public override Complex32 DotProduct(Complex32[] x, Complex32[] y) + { + if (y == null) + { + throw new ArgumentNullException("y"); + } + + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (x.Length != y.Length) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength); + } + + return SafeNativeMethods.c_dot_product(_blasHandle, x.Length, x, y); + } + + /// + /// Adds a scaled vector to another: result = y + alpha*x. + /// + /// The vector to update. + /// The value to scale by. + /// The vector to add to . + /// The result of the addition. + /// This is similar to the AXPY BLAS routine. + [SecuritySafeCritical] + public override void AddVectorToScaledVector(Complex32[] y, Complex32 alpha, Complex32[] x, Complex32[] result) + { + if (y == null) + { + throw new ArgumentNullException("y"); + } + + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (y.Length != x.Length) + { + throw new ArgumentException(Resources.ArgumentVectorsSameLength); + } + + if (!ReferenceEquals(y, result)) + { + Array.Copy(y, 0, result, 0, y.Length); + } + + if (alpha == Complex32.Zero) + { + return; + } + + SafeNativeMethods.c_axpy(_blasHandle, y.Length, alpha, x, result); + } + + /// + /// Scales an array. Can be used to scale a vector and a matrix. + /// + /// The scalar. + /// The values to scale. + /// This result of the scaling. + /// This is similar to the SCAL BLAS routine. + [SecuritySafeCritical] + public override void ScaleArray(Complex32 alpha, Complex32[] x, Complex32[] result) + { + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (!ReferenceEquals(x, result)) + { + Array.Copy(x, 0, result, 0, x.Length); + } + + if (alpha == Complex32.One) + { + return; + } + + SafeNativeMethods.c_scale(_blasHandle, x.Length, alpha, result); + } + + /// + /// Multiples two matrices. result = x * y + /// + /// The x matrix. + /// The number of rows in the x matrix. + /// The number of columns in the x matrix. + /// The y matrix. + /// The number of rows in the y matrix. + /// The number of columns in the y matrix. + /// Where to store the result of the multiplication. + /// This is a simplified version of the BLAS GEMM routine with alpha + /// set to Complex32.One and beta set to Complex32.Zero, and x and y are not transposed. + public override void MatrixMultiply(Complex32[] x, int rowsX, int columnsX, Complex32[] y, int rowsY, int columnsY, Complex32[] result) + { + MatrixMultiplyWithUpdate(Transpose.DontTranspose, Transpose.DontTranspose, Complex32.One, x, rowsX, columnsX, y, rowsY, columnsY, Complex32.Zero, result); + } + + /// + /// Multiplies two matrices and updates another with the result. c = alpha*op(a)*op(b) + beta*c + /// + /// How to transpose the matrix. + /// How to transpose the matrix. + /// The value to scale matrix. + /// The a matrix. + /// The number of rows in the matrix. + /// The number of columns in the matrix. + /// The b matrix + /// The number of rows in the matrix. + /// The number of columns in the matrix. + /// The value to scale the matrix. + /// The c matrix. + [SecuritySafeCritical] + public override void MatrixMultiplyWithUpdate(Transpose transposeA, Transpose transposeB, Complex32 alpha, Complex32[] a, int rowsA, int columnsA, Complex32[] b, int rowsB, int columnsB, Complex32 beta, Complex32[] c) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (c == null) + { + throw new ArgumentNullException("c"); + } + + var m = transposeA == Transpose.DontTranspose ? rowsA : columnsA; + var n = transposeB == Transpose.DontTranspose ? columnsB : rowsB; + var k = transposeA == Transpose.DontTranspose ? columnsA : rowsA; + var l = transposeB == Transpose.DontTranspose ? rowsB : columnsB; + + if (c.Length != m*n) + { + throw new ArgumentException(Resources.ArgumentMatrixDimensions); + } + + if (k != l) + { + throw new ArgumentException(Resources.ArgumentMatrixDimensions); + } + + SafeNativeMethods.c_matrix_multiply(_blasHandle, transposeA.ToCUDA(), transposeB.ToCUDA(), m, n, k, alpha, a, b, beta, c); + } + + /// + /// Computes the LUP factorization of A. P*A = L*U. + /// + /// An by matrix. The matrix is overwritten with the + /// the LU factorization on exit. The lower triangular factor L is stored in under the diagonal of (the diagonal is always Complex32.One + /// for the L factor). The upper triangular factor U is stored on and above the diagonal of . + /// The order of the square matrix . + /// On exit, it contains the pivot indices. The size of the array must be . + /// This is equivalent to the GETRF LAPACK routine. + [SecuritySafeCritical] + public override void LUFactor(Complex32[] data, int order, int[] ipiv) + { + if (data == null) + { + throw new ArgumentNullException("data"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (data.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "data"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + Solver(SafeNativeMethods.c_lu_factor(_solverHandle, order, data, ipiv)); + } + + /// + /// Computes the inverse of matrix using LU factorization. + /// + /// The N by N matrix to invert. Contains the inverse On exit. + /// The order of the square matrix . + /// This is equivalent to the GETRF and GETRI LAPACK routines. + [SecuritySafeCritical] + public override void LUInverse(Complex32[] a, int order) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + Solver(SafeNativeMethods.c_lu_inverse(_solverHandle, _blasHandle, order, a)); + } + + /// + /// Computes the inverse of a previously factored matrix. + /// + /// The LU factored N by N matrix. Contains the inverse On exit. + /// The order of the square matrix . + /// The pivot indices of . + /// This is equivalent to the GETRI LAPACK routine. + [SecuritySafeCritical] + public override void LUInverseFactored(Complex32[] a, int order, int[] ipiv) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + BLAS(SafeNativeMethods.c_lu_inverse_factored(_blasHandle, order, a, ipiv)); + } + + /// + /// Computes the inverse of matrix using LU factorization. + /// + /// The N by N matrix to invert. Contains the inverse On exit. + /// The order of the square matrix . + /// Not supported. Should be left null. + /// This is equivalent to the GETRF and GETRI LAPACK routines. + [SecuritySafeCritical] + public override void LUInverse(Complex32[] a, int order, Complex32[] work) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (work != null) + { + throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + } + + Solver(SafeNativeMethods.c_lu_inverse(_solverHandle, _blasHandle, order, a)); + } + + /// + /// Computes the inverse of a previously factored matrix. + /// + /// The LU factored N by N matrix. Contains the inverse On exit. + /// The order of the square matrix . + /// The pivot indices of . + /// Not supported. Should be left null. + /// This is equivalent to the GETRI LAPACK routine. + [SecuritySafeCritical] + public override void LUInverseFactored(Complex32[] a, int order, int[] ipiv, Complex32[] work) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + if (work != null) + { + throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + } + + BLAS(SafeNativeMethods.c_lu_inverse_factored(_blasHandle, order, a, ipiv)); + } + + /// + /// Solves A*X=B for X using LU factorization. + /// + /// The number of columns of B. + /// The square matrix A. + /// The order of the square matrix . + /// On entry the B matrix; on exit the X matrix. + /// This is equivalent to the GETRF and GETRS LAPACK routines. + [SecuritySafeCritical] + public override void LUSolve(int columnsOfB, Complex32[] a, int order, Complex32[] b) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (b.Length != columnsOfB*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.c_lu_solve(_solverHandle, order, columnsOfB, a, b)); + } + + /// + /// Solves A*X=B for X using a previously factored A matrix. + /// + /// The number of columns of B. + /// The factored A matrix. + /// The order of the square matrix . + /// The pivot indices of . + /// On entry the B matrix; on exit the X matrix. + /// This is equivalent to the GETRS LAPACK routine. + [SecuritySafeCritical] + public override void LUSolveFactored(int columnsOfB, Complex32[] a, int order, int[] ipiv, Complex32[] b) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + if (b.Length != columnsOfB*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.c_lu_solve_factored(_solverHandle, order, columnsOfB, a, ipiv, b)); + } + + /// + /// Computes the Cholesky factorization of A. + /// + /// On entry, a square, positive definite matrix. On exit, the matrix is overwritten with the + /// the Cholesky factorization. + /// The number of rows or columns in the matrix. + /// This is equivalent to the POTRF LAPACK routine. + [SecuritySafeCritical] + public override void CholeskyFactor(Complex32[] a, int order) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (order < 1) + { + throw new ArgumentException(Resources.ArgumentMustBePositive, "order"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + Solver(SafeNativeMethods.c_cholesky_factor(_solverHandle, order, a)); + } + + /// + /// Solves A*X=B for X using Cholesky factorization. + /// + /// The square, positive definite matrix A. + /// The number of rows and columns in A. + /// On entry the B matrix; on exit the X matrix. + /// The number of columns in the B matrix. + /// This is equivalent to the POTRF add POTRS LAPACK routines. + /// + [SecuritySafeCritical] + public override void CholeskySolve(Complex32[] a, int orderA, Complex32[] b, int columnsB) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (b.Length != orderA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.c_cholesky_solve(_solverHandle, orderA, columnsB, a, b)); + } + + /// + /// Solves A*X=B for X using a previously factored A matrix. + /// + /// The square, positive definite matrix A. + /// The number of rows and columns in A. + /// On entry the B matrix; on exit the X matrix. + /// The number of columns in the B matrix. + /// This is equivalent to the POTRS LAPACK routine. + [SecuritySafeCritical] + public override void CholeskySolveFactored(Complex32[] a, int orderA, Complex32[] b, int columnsB) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (b.Length != orderA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.c_cholesky_solve_factored(_solverHandle, orderA, columnsB, a, b)); + } + + /// + /// Computes the singular value decomposition of A. + /// + /// Compute the singular U and VT vectors or not. + /// On entry, the M by N matrix to decompose. On exit, A may be overwritten. + /// The number of rows in the A matrix. + /// The number of columns in the A matrix. + /// The singular values of A in ascending value. + /// If is true, on exit U contains the left + /// singular vectors. + /// If is true, on exit VT contains the transposed + /// right singular vectors. + /// This is equivalent to the GESVD LAPACK routine. + [SecuritySafeCritical] + public override void SingularValueDecomposition(bool computeVectors, Complex32[] a, int rowsA, int columnsA, Complex32[] s, Complex32[] u, Complex32[] vt) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (s == null) + { + throw new ArgumentNullException("s"); + } + + if (u == null) + { + throw new ArgumentNullException("u"); + } + + if (vt == null) + { + throw new ArgumentNullException("vt"); + } + + if (u.Length != rowsA*rowsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "u"); + } + + if (vt.Length != columnsA*columnsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "vt"); + } + + if (s.Length != Math.Min(rowsA, columnsA)) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "s"); + } + + SingularValueDecomposition(computeVectors, a, rowsA, columnsA, s, u, vt); + } + + /// + /// Solves A*X=B for X using the singular value decomposition of A. + /// + /// On entry, the M by N matrix to decompose. + /// The number of rows in the A matrix. + /// The number of columns in the A matrix. + /// The B matrix. + /// The number of columns of B. + /// On exit, the solution matrix. + public override void SvdSolve(Complex32[] a, int rowsA, int columnsA, Complex32[] b, int columnsB, Complex32[] x) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (b.Length != rowsA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (x.Length != columnsA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + var s = new Complex32[Math.Min(rowsA, columnsA)]; + var u = new Complex32[rowsA*rowsA]; + var vt = new Complex32[columnsA*columnsA]; + + var clone = new Complex32[a.Length]; + a.Copy(clone); + SingularValueDecomposition(true, clone, rowsA, columnsA, s, u, vt, null); + SvdSolveFactored(rowsA, columnsA, s, u, vt, b, columnsB, x); + } + + /// + /// Computes the singular value decomposition of A. + /// + /// Compute the singular U and VT vectors or not. + /// On entry, the M by N matrix to decompose. On exit, A may be overwritten. + /// The number of rows in the A matrix. + /// The number of columns in the A matrix. + /// The singular values of A in ascending value. + /// If is true, on exit U contains the left + /// singular vectors. + /// If is true, on exit VT contains the transposed + /// right singular vectors. + /// Not supported. Should be left null. + /// This is equivalent to the GESVD LAPACK routine. + [SecuritySafeCritical] + public override void SingularValueDecomposition(bool computeVectors, Complex32[] a, int rowsA, int columnsA, Complex32[] s, Complex32[] u, Complex32[] vt, Complex32[] work) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (s == null) + { + throw new ArgumentNullException("s"); + } + + if (u == null) + { + throw new ArgumentNullException("u"); + } + + if (vt == null) + { + throw new ArgumentNullException("vt"); + } + + if (work != null) + { + throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + } + + if (u.Length != rowsA*rowsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "u"); + } + + if (vt.Length != columnsA*columnsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "vt"); + } + + if (s.Length != Math.Min(rowsA, columnsA)) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "s"); + } + + Solver(SafeNativeMethods.c_svd_factor(_solverHandle, computeVectors, rowsA, columnsA, a, s, u, vt)); + } + } +} + +#endif diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Double.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Double.cs new file mode 100644 index 00000000..8213d8a5 --- /dev/null +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Double.cs @@ -0,0 +1,704 @@ +// +// Math.NET Numerics, part of the Math.NET Project +// http://numerics.mathdotnet.com +// http://github.com/mathnet/mathnet-numerics +// http://mathnetnumerics.codeplex.com +// +// Copyright (c) 2009-2013 Math.NET +// +// Permission is hereby granted, free of charge, to any person +// obtaining a copy of this software and associated documentation +// files (the "Software"), to deal in the Software without +// restriction, including without limitation the rights to use, +// copy, modify, merge, publish, distribute, sublicense, and/or sell +// copies of the Software, and to permit persons to whom the +// Software is furnished to do so, subject to the following +// conditions: +// +// The above copyright notice and this permission notice shall be +// included in all copies or substantial portions of the Software. +// +// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, +// EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES +// OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND +// NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT +// HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, +// WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +// FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR +// OTHER DEALINGS IN THE SOFTWARE. +// + +#if NATIVE + +using System; +using System.Numerics; +using System.Security; +using MathNet.Numerics.LinearAlgebra.Factorization; +using MathNet.Numerics.Properties; + +namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda +{ + /// + /// Intel's Math Kernel Library (MKL) linear algebra provider. + /// + public partial class CudaLinearAlgebraProvider + { + /// + /// Computes the dot product of x and y. + /// + /// The vector x. + /// The vector y. + /// The dot product of x and y. + /// This is equivalent to the DOT BLAS routine. + [SecuritySafeCritical] + public override double DotProduct(double[] x, double[] y) + { + if (y == null) + { + throw new ArgumentNullException("y"); + } + + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (x.Length != y.Length) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength); + } + + return SafeNativeMethods.d_dot_product(_blasHandle, x.Length, x, y); + } + + /// + /// Adds a scaled vector to another: result = y + alpha*x. + /// + /// The vector to update. + /// The value to scale by. + /// The vector to add to . + /// The result of the addition. + /// This is similar to the AXPY BLAS routine. + [SecuritySafeCritical] + public override void AddVectorToScaledVector(double[] y, double alpha, double[] x, double[] result) + { + if (y == null) + { + throw new ArgumentNullException("y"); + } + + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (y.Length != x.Length) + { + throw new ArgumentException(Resources.ArgumentVectorsSameLength); + } + + if (!ReferenceEquals(y, result)) + { + Array.Copy(y, 0, result, 0, y.Length); + } + + if (alpha == 0.0) + { + return; + } + + SafeNativeMethods.d_axpy(_blasHandle, y.Length, alpha, x, result); + } + + /// + /// Scales an array. Can be used to scale a vector and a matrix. + /// + /// The scalar. + /// The values to scale. + /// This result of the scaling. + /// This is similar to the SCAL BLAS routine. + [SecuritySafeCritical] + public override void ScaleArray(double alpha, double[] x, double[] result) + { + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (!ReferenceEquals(x, result)) + { + Array.Copy(x, 0, result, 0, x.Length); + } + + if (alpha == 1.0) + { + return; + } + + SafeNativeMethods.d_scale(_blasHandle, x.Length, alpha, result); + } + + /// + /// Multiples two matrices. result = x * y + /// + /// The x matrix. + /// The number of rows in the x matrix. + /// The number of columns in the x matrix. + /// The y matrix. + /// The number of rows in the y matrix. + /// The number of columns in the y matrix. + /// Where to store the result of the multiplication. + /// This is a simplified version of the BLAS GEMM routine with alpha + /// set to 1.0 and beta set to 0.0, and x and y are not transposed. + public override void MatrixMultiply(double[] x, int rowsX, int columnsX, double[] y, int rowsY, int columnsY, double[] result) + { + MatrixMultiplyWithUpdate(Transpose.DontTranspose, Transpose.DontTranspose, 1.0, x, rowsX, columnsX, y, rowsY, columnsY, 0.0, result); + } + + /// + /// Multiplies two matrices and updates another with the result. c = alpha*op(a)*op(b) + beta*c + /// + /// How to transpose the matrix. + /// How to transpose the matrix. + /// The value to scale matrix. + /// The a matrix. + /// The number of rows in the matrix. + /// The number of columns in the matrix. + /// The b matrix + /// The number of rows in the matrix. + /// The number of columns in the matrix. + /// The value to scale the matrix. + /// The c matrix. + [SecuritySafeCritical] + public override void MatrixMultiplyWithUpdate(Transpose transposeA, Transpose transposeB, double alpha, double[] a, int rowsA, int columnsA, double[] b, int rowsB, int columnsB, double beta, double[] c) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (c == null) + { + throw new ArgumentNullException("c"); + } + + var m = transposeA == Transpose.DontTranspose ? rowsA : columnsA; + var n = transposeB == Transpose.DontTranspose ? columnsB : rowsB; + var k = transposeA == Transpose.DontTranspose ? columnsA : rowsA; + var l = transposeB == Transpose.DontTranspose ? rowsB : columnsB; + + if (c.Length != m*n) + { + throw new ArgumentException(Resources.ArgumentMatrixDimensions); + } + + if (k != l) + { + throw new ArgumentException(Resources.ArgumentMatrixDimensions); + } + + SafeNativeMethods.d_matrix_multiply(_blasHandle, transposeA.ToCUDA(), transposeB.ToCUDA(), m, n, k, alpha, a, b, beta, c); + } + + /// + /// Computes the LUP factorization of A. P*A = L*U. + /// + /// An by matrix. The matrix is overwritten with the + /// the LU factorization on exit. The lower triangular factor L is stored in under the diagonal of (the diagonal is always 1.0 + /// for the L factor). The upper triangular factor U is stored on and above the diagonal of . + /// The order of the square matrix . + /// On exit, it contains the pivot indices. The size of the array must be . + /// This is equivalent to the GETRF LAPACK routine. + [SecuritySafeCritical] + public override void LUFactor(double[] data, int order, int[] ipiv) + { + if (data == null) + { + throw new ArgumentNullException("data"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (data.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "data"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + Solver(SafeNativeMethods.d_lu_factor(_solverHandle, order, data, ipiv)); + } + + /// + /// Computes the inverse of matrix using LU factorization. + /// + /// The N by N matrix to invert. Contains the inverse On exit. + /// The order of the square matrix . + /// This is equivalent to the GETRF and GETRI LAPACK routines. + [SecuritySafeCritical] + public override void LUInverse(double[] a, int order) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + Solver(SafeNativeMethods.d_lu_inverse(_solverHandle, _blasHandle, order, a)); + } + + /// + /// Computes the inverse of a previously factored matrix. + /// + /// The LU factored N by N matrix. Contains the inverse On exit. + /// The order of the square matrix . + /// The pivot indices of . + /// This is equivalent to the GETRI LAPACK routine. + [SecuritySafeCritical] + public override void LUInverseFactored(double[] a, int order, int[] ipiv) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + BLAS(SafeNativeMethods.d_lu_inverse_factored(_blasHandle, order, a, ipiv)); + } + + /// + /// Computes the inverse of matrix using LU factorization. + /// + /// The N by N matrix to invert. Contains the inverse On exit. + /// The order of the square matrix . + /// Not supported. Should be left null. + /// This is equivalent to the GETRF and GETRI LAPACK routines. + [SecuritySafeCritical] + public override void LUInverse(double[] a, int order, double[] work) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (work != null) + { + throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + } + + Solver(SafeNativeMethods.d_lu_inverse(_solverHandle, _blasHandle, order, a)); + } + + /// + /// Computes the inverse of a previously factored matrix. + /// + /// The LU factored N by N matrix. Contains the inverse On exit. + /// The order of the square matrix . + /// The pivot indices of . + /// Not supported. Should be left null. + /// This is equivalent to the GETRI LAPACK routine. + [SecuritySafeCritical] + public override void LUInverseFactored(double[] a, int order, int[] ipiv, double[] work) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + if (work != null) + { + throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + } + + BLAS(SafeNativeMethods.d_lu_inverse_factored(_blasHandle, order, a, ipiv)); + } + + /// + /// Solves A*X=B for X using LU factorization. + /// + /// The number of columns of B. + /// The square matrix A. + /// The order of the square matrix . + /// On entry the B matrix; on exit the X matrix. + /// This is equivalent to the GETRF and GETRS LAPACK routines. + [SecuritySafeCritical] + public override void LUSolve(int columnsOfB, double[] a, int order, double[] b) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (b.Length != columnsOfB*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.d_lu_solve(_solverHandle, order, columnsOfB, a, b)); + } + + /// + /// Solves A*X=B for X using a previously factored A matrix. + /// + /// The number of columns of B. + /// The factored A matrix. + /// The order of the square matrix . + /// The pivot indices of . + /// On entry the B matrix; on exit the X matrix. + /// This is equivalent to the GETRS LAPACK routine. + [SecuritySafeCritical] + public override void LUSolveFactored(int columnsOfB, double[] a, int order, int[] ipiv, double[] b) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + if (b.Length != columnsOfB*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.d_lu_solve_factored(_solverHandle, order, columnsOfB, a, ipiv, b)); + } + + /// + /// Computes the Cholesky factorization of A. + /// + /// On entry, a square, positive definite matrix. On exit, the matrix is overwritten with the + /// the Cholesky factorization. + /// The number of rows or columns in the matrix. + /// This is equivalent to the POTRF LAPACK routine. + [SecuritySafeCritical] + public override void CholeskyFactor(double[] a, int order) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (order < 1) + { + throw new ArgumentException(Resources.ArgumentMustBePositive, "order"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + Solver(SafeNativeMethods.d_cholesky_factor(_solverHandle, order, a)); + } + + /// + /// Solves A*X=B for X using Cholesky factorization. + /// + /// The square, positive definite matrix A. + /// The number of rows and columns in A. + /// On entry the B matrix; on exit the X matrix. + /// The number of columns in the B matrix. + /// This is equivalent to the POTRF add POTRS LAPACK routines. + /// + [SecuritySafeCritical] + public override void CholeskySolve(double[] a, int orderA, double[] b, int columnsB) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (b.Length != orderA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.d_cholesky_solve(_solverHandle, orderA, columnsB, a, b)); + } + + /// + /// Solves A*X=B for X using a previously factored A matrix. + /// + /// The square, positive definite matrix A. + /// The number of rows and columns in A. + /// On entry the B matrix; on exit the X matrix. + /// The number of columns in the B matrix. + /// This is equivalent to the POTRS LAPACK routine. + [SecuritySafeCritical] + public override void CholeskySolveFactored(double[] a, int orderA, double[] b, int columnsB) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (b.Length != orderA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.d_cholesky_solve_factored(_solverHandle, orderA, columnsB, a, b)); + } + + /// + /// Computes the singular value decomposition of A. + /// + /// Compute the singular U and VT vectors or not. + /// On entry, the M by N matrix to decompose. On exit, A may be overwritten. + /// The number of rows in the A matrix. + /// The number of columns in the A matrix. + /// The singular values of A in ascending value. + /// If is true, on exit U contains the left + /// singular vectors. + /// If is true, on exit VT contains the transposed + /// right singular vectors. + /// This is equivalent to the GESVD LAPACK routine. + [SecuritySafeCritical] + public override void SingularValueDecomposition(bool computeVectors, double[] a, int rowsA, int columnsA, double[] s, double[] u, double[] vt) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (s == null) + { + throw new ArgumentNullException("s"); + } + + if (u == null) + { + throw new ArgumentNullException("u"); + } + + if (vt == null) + { + throw new ArgumentNullException("vt"); + } + + if (u.Length != rowsA*rowsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "u"); + } + + if (vt.Length != columnsA*columnsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "vt"); + } + + if (s.Length != Math.Min(rowsA, columnsA)) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "s"); + } + + SingularValueDecomposition(computeVectors, a, rowsA, columnsA, s, u, vt); + } + + /// + /// Solves A*X=B for X using the singular value decomposition of A. + /// + /// On entry, the M by N matrix to decompose. + /// The number of rows in the A matrix. + /// The number of columns in the A matrix. + /// The B matrix. + /// The number of columns of B. + /// On exit, the solution matrix. + public override void SvdSolve(double[] a, int rowsA, int columnsA, double[] b, int columnsB, double[] x) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (b.Length != rowsA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (x.Length != columnsA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + var s = new double[Math.Min(rowsA, columnsA)]; + var u = new double[rowsA*rowsA]; + var vt = new double[columnsA*columnsA]; + + var clone = new double[a.Length]; + a.Copy(clone); + SingularValueDecomposition(true, clone, rowsA, columnsA, s, u, vt); + SvdSolveFactored(rowsA, columnsA, s, u, vt, b, columnsB, x); + } + + /// + /// Computes the singular value decomposition of A. + /// + /// Compute the singular U and VT vectors or not. + /// On entry, the M by N matrix to decompose. On exit, A may be overwritten. + /// The number of rows in the A matrix. + /// The number of columns in the A matrix. + /// The singular values of A in ascending value. + /// If is true, on exit U contains the left + /// singular vectors. + /// If is true, on exit VT contains the transposed + /// right singular vectors. + /// Not supported. Should be left null. + /// This is equivalent to the GESVD LAPACK routine. + [SecuritySafeCritical] + public override void SingularValueDecomposition(bool computeVectors, double[] a, int rowsA, int columnsA, double[] s, double[] u, double[] vt, double[] work) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (s == null) + { + throw new ArgumentNullException("s"); + } + + if (u == null) + { + throw new ArgumentNullException("u"); + } + + if (vt == null) + { + throw new ArgumentNullException("vt"); + } + + if (work != null) + { + throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + } + + if (u.Length != rowsA*rowsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "u"); + } + + if (vt.Length != columnsA*columnsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "vt"); + } + + if (s.Length != Math.Min(rowsA, columnsA)) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "s"); + } + + + Solver (SafeNativeMethods.d_svd_factor(_solverHandle, computeVectors, rowsA, columnsA, a, s, u, vt)); + } + } +} + +#endif diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Single.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Single.cs new file mode 100644 index 00000000..f964e286 --- /dev/null +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Single.cs @@ -0,0 +1,703 @@ +// +// Math.NET Numerics, part of the Math.NET Project +// http://numerics.mathdotnet.com +// http://github.com/mathnet/mathnet-numerics +// http://mathnetnumerics.codeplex.com +// +// Copyright (c) 2009-2013 Math.NET +// +// Permission is hereby granted, free of charge, to any person +// obtaining a copy of this software and associated documentation +// files (the "Software"), to deal in the Software without +// restriction, including without limitation the rights to use, +// copy, modify, merge, publish, distribute, sublicense, and/or sell +// copies of the Software, and to permit persons to whom the +// Software is furnished to do so, subject to the following +// conditions: +// +// The above copyright notice and this permission notice shall be +// included in all copies or substantial portions of the Software. +// +// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, +// EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES +// OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND +// NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT +// HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, +// WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +// FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR +// OTHER DEALINGS IN THE SOFTWARE. +// + +#if NATIVE + +using System; +using System.Numerics; +using System.Security; +using MathNet.Numerics.LinearAlgebra.Factorization; +using MathNet.Numerics.Properties; + +namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda +{ + /// + /// Intel's Math Kernel Library (MKL) linear algebra provider. + /// + public partial class CudaLinearAlgebraProvider + { + /// + /// Computes the dot product of x and y. + /// + /// The vector x. + /// The vector y. + /// The dot product of x and y. + /// This is equivalent to the DOT BLAS routine. + [SecuritySafeCritical] + public override float DotProduct(float[] x, float[] y) + { + if (y == null) + { + throw new ArgumentNullException("y"); + } + + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (x.Length != y.Length) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength); + } + + return SafeNativeMethods.s_dot_product(_blasHandle, x.Length, x, y); + } + + /// + /// Adds a scaled vector to another: result = y + alpha*x. + /// + /// The vector to update. + /// The value to scale by. + /// The vector to add to . + /// The result of the addition. + /// This is similar to the AXPY BLAS routine. + [SecuritySafeCritical] + public override void AddVectorToScaledVector(float[] y, float alpha, float[] x, float[] result) + { + if (y == null) + { + throw new ArgumentNullException("y"); + } + + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (y.Length != x.Length) + { + throw new ArgumentException(Resources.ArgumentVectorsSameLength); + } + + if (!ReferenceEquals(y, result)) + { + Array.Copy(y, 0, result, 0, y.Length); + } + + if (alpha == 0.0f) + { + return; + } + + SafeNativeMethods.s_axpy(_blasHandle, y.Length, alpha, x, result); + } + + /// + /// Scales an array. Can be used to scale a vector and a matrix. + /// + /// The scalar. + /// The values to scale. + /// This result of the scaling. + /// This is similar to the SCAL BLAS routine. + [SecuritySafeCritical] + public override void ScaleArray(float alpha, float[] x, float[] result) + { + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (!ReferenceEquals(x, result)) + { + Array.Copy(x, 0, result, 0, x.Length); + } + + if (alpha == 1.0f) + { + return; + } + + SafeNativeMethods.s_scale(_blasHandle, x.Length, alpha, result); + } + + /// + /// Multiples two matrices. result = x * y + /// + /// The x matrix. + /// The number of rows in the x matrix. + /// The number of columns in the x matrix. + /// The y matrix. + /// The number of rows in the y matrix. + /// The number of columns in the y matrix. + /// Where to store the result of the multiplication. + /// This is a simplified version of the BLAS GEMM routine with alpha + /// set to 1.0f and beta set to 0.0f, and x and y are not transposed. + public override void MatrixMultiply(float[] x, int rowsX, int columnsX, float[] y, int rowsY, int columnsY, float[] result) + { + MatrixMultiplyWithUpdate(Transpose.DontTranspose, Transpose.DontTranspose, 1.0f, x, rowsX, columnsX, y, rowsY, columnsY, 0.0f, result); + } + + /// + /// Multiplies two matrices and updates another with the result. c = alpha*op(a)*op(b) + beta*c + /// + /// How to transpose the matrix. + /// How to transpose the matrix. + /// The value to scale matrix. + /// The a matrix. + /// The number of rows in the matrix. + /// The number of columns in the matrix. + /// The b matrix + /// The number of rows in the matrix. + /// The number of columns in the matrix. + /// The value to scale the matrix. + /// The c matrix. + [SecuritySafeCritical] + public override void MatrixMultiplyWithUpdate(Transpose transposeA, Transpose transposeB, float alpha, float[] a, int rowsA, int columnsA, float[] b, int rowsB, int columnsB, float beta, float[] c) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (c == null) + { + throw new ArgumentNullException("c"); + } + + var m = transposeA == Transpose.DontTranspose ? rowsA : columnsA; + var n = transposeB == Transpose.DontTranspose ? columnsB : rowsB; + var k = transposeA == Transpose.DontTranspose ? columnsA : rowsA; + var l = transposeB == Transpose.DontTranspose ? rowsB : columnsB; + + if (c.Length != m*n) + { + throw new ArgumentException(Resources.ArgumentMatrixDimensions); + } + + if (k != l) + { + throw new ArgumentException(Resources.ArgumentMatrixDimensions); + } + + SafeNativeMethods.s_matrix_multiply(_blasHandle, transposeA.ToCUDA(), transposeB.ToCUDA(), m, n, k, alpha, a, b, beta, c); + } + + /// + /// Computes the LUP factorization of A. P*A = L*U. + /// + /// An by matrix. The matrix is overwritten with the + /// the LU factorization on exit. The lower triangular factor L is stored in under the diagonal of (the diagonal is always 1.0f + /// for the L factor). The upper triangular factor U is stored on and above the diagonal of . + /// The order of the square matrix . + /// On exit, it contains the pivot indices. The size of the array must be . + /// This is equivalent to the GETRF LAPACK routine. + [SecuritySafeCritical] + public override void LUFactor(float[] data, int order, int[] ipiv) + { + if (data == null) + { + throw new ArgumentNullException("data"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (data.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "data"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + Solver(SafeNativeMethods.s_lu_factor(_solverHandle, order, data, ipiv)); + } + + /// + /// Computes the inverse of matrix using LU factorization. + /// + /// The N by N matrix to invert. Contains the inverse On exit. + /// The order of the square matrix . + /// This is equivalent to the GETRF and GETRI LAPACK routines. + [SecuritySafeCritical] + public override void LUInverse(float[] a, int order) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + Solver(SafeNativeMethods.s_lu_inverse(_solverHandle, _blasHandle, order, a)); + } + + /// + /// Computes the inverse of a previously factored matrix. + /// + /// The LU factored N by N matrix. Contains the inverse On exit. + /// The order of the square matrix . + /// The pivot indices of . + /// This is equivalent to the GETRI LAPACK routine. + [SecuritySafeCritical] + public override void LUInverseFactored(float[] a, int order, int[] ipiv) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + BLAS(SafeNativeMethods.s_lu_inverse_factored(_blasHandle, order, a, ipiv)); + } + + /// + /// Computes the inverse of matrix using LU factorization. + /// + /// The N by N matrix to invert. Contains the inverse On exit. + /// The order of the square matrix . + /// Not supported. Should be left null. + /// This is equivalent to the GETRF and GETRI LAPACK routines. + [SecuritySafeCritical] + public override void LUInverse(float[] a, int order, float[] work) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (work != null) + { + throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + } + + Solver(SafeNativeMethods.s_lu_inverse(_solverHandle, _blasHandle, order, a)); + } + + /// + /// Computes the inverse of a previously factored matrix. + /// + /// The LU factored N by N matrix. Contains the inverse On exit. + /// The order of the square matrix . + /// The pivot indices of . + /// Not supported. This should be left null. + /// This is equivalent to the GETRI LAPACK routine. + [SecuritySafeCritical] + public override void LUInverseFactored(float[] a, int order, int[] ipiv, float[] work) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + if (work != null) + { + throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + } + + BLAS(SafeNativeMethods.s_lu_inverse_factored(_blasHandle, order, a, ipiv)); + } + + /// + /// Solves A*X=B for X using LU factorization. + /// + /// The number of columns of B. + /// The square matrix A. + /// The order of the square matrix . + /// On entry the B matrix; on exit the X matrix. + /// This is equivalent to the GETRF and GETRS LAPACK routines. + [SecuritySafeCritical] + public override void LUSolve(int columnsOfB, float[] a, int order, float[] b) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (b.Length != columnsOfB*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.s_lu_solve(_solverHandle, order, columnsOfB, a, b)); + } + + /// + /// Solves A*X=B for X using a previously factored A matrix. + /// + /// The number of columns of B. + /// The factored A matrix. + /// The order of the square matrix . + /// The pivot indices of . + /// On entry the B matrix; on exit the X matrix. + /// This is equivalent to the GETRS LAPACK routine. + [SecuritySafeCritical] + public override void LUSolveFactored(int columnsOfB, float[] a, int order, int[] ipiv, float[] b) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (ipiv == null) + { + throw new ArgumentNullException("ipiv"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + if (ipiv.Length != order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "ipiv"); + } + + if (b.Length != columnsOfB*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.s_lu_solve_factored(_solverHandle, order, columnsOfB, a, ipiv, b)); + } + + /// + /// Computes the Cholesky factorization of A. + /// + /// On entry, a square, positive definite matrix. On exit, the matrix is overwritten with the + /// the Cholesky factorization. + /// The number of rows or columns in the matrix. + /// This is equivalent to the POTRF LAPACK routine. + [SecuritySafeCritical] + public override void CholeskyFactor(float[] a, int order) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (order < 1) + { + throw new ArgumentException(Resources.ArgumentMustBePositive, "order"); + } + + if (a.Length != order*order) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "a"); + } + + Solver(SafeNativeMethods.s_cholesky_factor(_solverHandle, order, a)); + } + + /// + /// Solves A*X=B for X using Cholesky factorization. + /// + /// The square, positive definite matrix A. + /// The number of rows and columns in A. + /// On entry the B matrix; on exit the X matrix. + /// The number of columns in the B matrix. + /// This is equivalent to the POTRF add POTRS LAPACK routines. + /// + [SecuritySafeCritical] + public override void CholeskySolve(float[] a, int orderA, float[] b, int columnsB) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (b.Length != orderA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.s_cholesky_solve(_solverHandle, orderA, columnsB, a, b)); + } + + /// + /// Solves A*X=B for X using a previously factored A matrix. + /// + /// The square, positive definite matrix A. + /// The number of rows and columns in A. + /// On entry the B matrix; on exit the X matrix. + /// The number of columns in the B matrix. + /// This is equivalent to the POTRS LAPACK routine. + [SecuritySafeCritical] + public override void CholeskySolveFactored(float[] a, int orderA, float[] b, int columnsB) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (b.Length != orderA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (ReferenceEquals(a, b)) + { + throw new ArgumentException(Resources.ArgumentReferenceDifferent); + } + + Solver(SafeNativeMethods.s_cholesky_solve_factored(_solverHandle, orderA, columnsB, a, b)); + } + + /// + /// Computes the singular value decomposition of A. + /// + /// Compute the singular U and VT vectors or not. + /// On entry, the M by N matrix to decompose. On exit, A may be overwritten. + /// The number of rows in the A matrix. + /// The number of columns in the A matrix. + /// The singular values of A in ascending value. + /// If is true, on exit U contains the left + /// singular vectors. + /// If is true, on exit VT contains the transposed + /// right singular vectors. + /// This is equivalent to the GESVD LAPACK routine. + [SecuritySafeCritical] + public override void SingularValueDecomposition(bool computeVectors, float[] a, int rowsA, int columnsA, float[] s, float[] u, float[] vt) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (s == null) + { + throw new ArgumentNullException("s"); + } + + if (u == null) + { + throw new ArgumentNullException("u"); + } + + if (vt == null) + { + throw new ArgumentNullException("vt"); + } + + if (u.Length != rowsA*rowsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "u"); + } + + if (vt.Length != columnsA*columnsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "vt"); + } + + if (s.Length != Math.Min(rowsA, columnsA)) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "s"); + } + + SingularValueDecomposition(computeVectors, a, rowsA, columnsA, s, u, vt, null); + } + + /// + /// Solves A*X=B for X using the singular value decomposition of A. + /// + /// On entry, the M by N matrix to decompose. + /// The number of rows in the A matrix. + /// The number of columns in the A matrix. + /// The B matrix. + /// The number of columns of B. + /// On exit, the solution matrix. + public override void SvdSolve(float[] a, int rowsA, int columnsA, float[] b, int columnsB, float[] x) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (b == null) + { + throw new ArgumentNullException("b"); + } + + if (x == null) + { + throw new ArgumentNullException("x"); + } + + if (b.Length != rowsA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + if (x.Length != columnsA*columnsB) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "b"); + } + + var s = new float[Math.Min(rowsA, columnsA)]; + var u = new float[rowsA*rowsA]; + var vt = new float[columnsA*columnsA]; + + var clone = new float[a.Length]; + a.Copy(clone); + SingularValueDecomposition(true, clone, rowsA, columnsA, s, u, vt, null); + SvdSolveFactored(rowsA, columnsA, s, u, vt, b, columnsB, x); + } + + /// + /// Computes the singular value decomposition of A. + /// + /// Compute the singular U and VT vectors or not. + /// On entry, the M by N matrix to decompose. On exit, A may be overwritten. + /// The number of rows in the A matrix. + /// The number of columns in the A matrix. + /// The singular values of A in ascending value. + /// If is true, on exit U contains the left + /// singular vectors. + /// If is true, on exit VT contains the transposed + /// right singular vectors. + /// Not supported. Should be left null. + /// This is equivalent to the GESVD LAPACK routine. + [SecuritySafeCritical] + public override void SingularValueDecomposition(bool computeVectors, float[] a, int rowsA, int columnsA, float[] s, float[] u, float[] vt, float[] work) + { + if (a == null) + { + throw new ArgumentNullException("a"); + } + + if (s == null) + { + throw new ArgumentNullException("s"); + } + + if (u == null) + { + throw new ArgumentNullException("u"); + } + + if (vt == null) + { + throw new ArgumentNullException("vt"); + } + + if (work != null) + { + throw new ArgumentException(Resources.UserWorkBufferNotSupported); + } + + if (u.Length != rowsA*rowsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "u"); + } + + if (vt.Length != columnsA*columnsA) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "vt"); + } + + if (s.Length != Math.Min(rowsA, columnsA)) + { + throw new ArgumentException(Resources.ArgumentArraysSameLength, "s"); + } + + Solver(SafeNativeMethods.s_svd_factor(_solverHandle, computeVectors, rowsA, columnsA, a, s, u, vt)); + } + } +} + +#endif diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs new file mode 100644 index 00000000..fe45efa8 --- /dev/null +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs @@ -0,0 +1,209 @@ +// +// Math.NET Numerics, part of the Math.NET Project +// http://numerics.mathdotnet.com +// http://github.com/mathnet/mathnet-numerics +// http://mathnetnumerics.codeplex.com +// +// Copyright (c) 2009-2015 Math.NET +// +// Permission is hereby granted, free of charge, to any person +// obtaining a copy of this software and associated documentation +// files (the "Software"), to deal in the Software without +// restriction, including without limitation the rights to use, +// copy, modify, merge, publish, distribute, sublicense, and/or sell +// copies of the Software, and to permit persons to whom the +// Software is furnished to do so, subject to the following +// conditions: +// +// The above copyright notice and this permission notice shall be +// included in all copies or substantial portions of the Software. +// +// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, +// EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES +// OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND +// NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT +// HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, +// WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +// FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR +// OTHER DEALINGS IN THE SOFTWARE. +// + +using System; + +#if NATIVE + +namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda +{ + /// + /// Consistency vs. performance trade-off between runs on different machines. + /// + + + /// + /// Intel's Math Kernel Library (MKL) linear algebra provider. + /// + public partial class CudaLinearAlgebraProvider : ManagedLinearAlgebraProvider, IDisposable + { + private int _nativeRevision; + private bool _nativeIX86; + private bool _nativeX64; + private bool _nativeIA64; + private IntPtr _blasHandle; + private IntPtr _solverHandle; + + /// + /// Sets the desired bit consistency on repeated identical computations on varying CPU architectures, + /// as a trade-off with performance. + /// + /// VML optimal precision and rounding. + /// VML accuracy mode. + [CLSCompliant(false)] + public CudaLinearAlgebraProvider() + { + } + + /// + /// Initialize and verify that the provided is indeed available. + /// If calling this method fails, consider to fall back to alternatives like the managed provider. + /// + public override void InitializeVerify() + { + int a, b, linearAlgebra; + try + { + // Load the native library + NativeProviderLoader.TryLoad(SafeNativeMethods.DllName); + + a = SafeNativeMethods.query_capability(0); + b = SafeNativeMethods.query_capability(1); + + _nativeIX86 = SafeNativeMethods.query_capability(8) > 0; + _nativeX64 = SafeNativeMethods.query_capability(9) > 0; + _nativeIA64 = SafeNativeMethods.query_capability(10) > 0; + + _nativeRevision = SafeNativeMethods.query_capability(64); + linearAlgebra = SafeNativeMethods.query_capability(128); + } + catch (DllNotFoundException e) + { + throw new NotSupportedException("Cuda Native Provider not found.", e); + } + catch (BadImageFormatException e) + { + throw new NotSupportedException("Cuda Native Provider found but failed to load. Please verify that the platform matches (x64 vs x32, Windows vs Linux).", e); + } + catch (EntryPointNotFoundException e) + { + throw new NotSupportedException("Cuda Native Provider does not support capability querying and is therefore not compatible. Consider upgrading to a newer version.", e); + } + + if (a != 0 || b != -1 || linearAlgebra <=0 || _nativeRevision < 1) + { + throw new NotSupportedException("Cuda Native Provider too old or not compatible. Consider upgrading to a newer version."); + } + + BLAS(SafeNativeMethods.createBLASHandle(ref _blasHandle)); + Solver(SafeNativeMethods.createSolverHandle(ref _solverHandle)); + } + + private void BLAS(int status) + { + switch (status) + { + case 0: // CUBLAS_STATUS_SUCCESS + return; + + case 1: // CUBLAS_STATUS_NOT_INITIALIZED + throw new Exception("The CUDA Runtime initialization failed"); + + case 2: // CUSOLVER_STATUS_ALLOC_FAILED + throw new OutOfMemoryException("The resources could not be allocated"); + + case 7: // CUBLAS_STATUS_INVALID_VALUE + throw new ArgumentException("Invalid value"); + + case 8: // CUBLAS_STATUS_ARCH_MISMATCH + throw new NotSupportedException("The device does not support this opeation."); + + case 11: // CUBLAS_STATUS_MAPPING_ERROR + throw new Exception("Mapping error."); + + case 13: // CUBLAS_STATUS_EXECUTION_FAILED + throw new Exception("Execution failed"); + + case 14: // CUBLAS_STATUS_INTERNAL_ERROR + throw new Exception("Internal error"); + + case 15: // CUBLAS_STATUS_NOT_SUPPORTED + throw new NotSupportedException(); + + case 16: // CUBLAS_STATUS_LICENSE_ERROR + throw new Exception("License error"); + + default: + throw new Exception("Unrecognized cuBLAS status code: " + status); + } + } + + private void Solver(int status) + { + switch (status) + { + case 0: // CUSOLVER_STATUS_SUCCESS + return; + + case 1: // CUSOLVER_STATUS_NOT_INITIALIZED + throw new Exception("The library was not initialized"); + + case 2: // CUSOLVER_STATUS_ALLOC_FAILED + throw new OutOfMemoryException("The resources could not be allocated"); + + case 3: // CUSOLVER_STATUS_INVALID_VALUE + throw new ArgumentException("Invalid value"); + + case 4: // CUSOLVER_STATUS_ARCH_MISMATCH + throw new NotSupportedException("The device does not support compute capability 2.0 and above"); + + case 5: // CUSOLVER_STATUS_MAPPING_ERROR + throw new Exception("Mapping error"); + + case 6: // CUSOLVER_STATUS_EXECUTION_FAILED + throw new Exception("Execution failed"); + + case 7: //CUSOLVER_STATUS_INTERNAL_ERROR + throw new Exception("Internal error"); + + case 8: // CUSOLVER_STATUS_MATRIX_TYPE_NOT_SUPPORTED + throw new NotSupportedException("Matrix type not supported"); + + case 9: // CUSOLVER_STATUS_NOT_SUPPORTED + throw new NotSupportedException(); + + case 10: // CUSOLVER_STATUS_ZERO_PIVOT + throw new Exception("Zero pivot"); + + case 11: //CUSOLVER_STATUS_INVALID_LICENSE + throw new Exception("Invalid license"); + + default: + throw new Exception("Unrecognized cuSolverDn status code: " + status); + + + } + } + + public override string ToString() + { + return string.Format("Nvidia CUDA ({1}; revision {0})", _nativeRevision, _nativeIX86 ? "x86" : _nativeX64 ? "x64" : _nativeIA64 ? "IA64" : "unknown"); + } + + + public void Dispose() + { + BLAS(SafeNativeMethods.destroyBLASHandle(_blasHandle)); + Solver(SafeNativeMethods.destroySolverHandle(_solverHandle)); + } + } +} + +#endif diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/SafeNativeMethods.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/SafeNativeMethods.cs new file mode 100644 index 00000000..ba58a483 --- /dev/null +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/SafeNativeMethods.cs @@ -0,0 +1,378 @@ +// +// Math.NET Numerics, part of the Math.NET Project +// http://mathnet.opensourcedotnet.info +// +// Copyright (c) 2009-2014 Math.NET +// +// Permission is hereby granted, free of charge, to any person +// obtaining a copy of this software and associated documentation +// files (the "Software"), to deal in the Software without +// restriction, including without limitation the rights to use, +// copy, modify, merge, publish, distribute, sublicense, and/or sell +// copies of the Software, and to permit persons to whom the +// Software is furnished to do so, subject to the following +// conditions: +// +// The above copyright notice and this permission notice shall be +// included in all copies or substantial portions of the Software. +// +// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, +// EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES +// OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND +// NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT +// HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, +// WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +// FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR +// OTHER DEALINGS IN THE SOFTWARE. +// + +#if NATIVE + +using System; +using System.Numerics; +using System.Runtime.InteropServices; +using System.Security; + +namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda +{ + /// + /// P/Invoke methods to the native math libraries. + /// + [SuppressUnmanagedCodeSecurity] + [SecurityCritical] + internal static class SafeNativeMethods + { + // ReSharper disable InconsistentNaming + + /// + /// Name of the native DLL. + /// + const string _DllName = "MathNet.Numerics.CUDA.dll"; + internal static string DllName { get { return _DllName; } } + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int query_capability(int capability); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int createBLASHandle(ref IntPtr blasHandle); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int destroyBLASHandle(IntPtr blasHandle); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int createSolverHandle(ref IntPtr solverHandle); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int destroySolverHandle(IntPtr solverHandle); + + #region BLAS + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern void s_axpy(IntPtr blasHandle, int n, float alpha, float[] x, [In, Out] float[] y); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern void d_axpy(IntPtr blasHandle, int n, double alpha, double[] x, [In, Out] double[] y); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern void c_axpy(IntPtr blasHandle, int n, Complex32 alpha, Complex32[] x, [In, Out] Complex32[] y); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern void z_axpy(IntPtr blasHandle, int n, Complex alpha, Complex[] x, [In, Out] Complex[] y); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern void s_scale(IntPtr blasHandle, int n, float alpha, [Out] float[] x); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern void d_scale(IntPtr blasHandle, int n, double alpha, [Out] double[] x); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern void c_scale(IntPtr blasHandle, int n, Complex32 alpha, [In, Out] Complex32[] x); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern void z_scale(IntPtr blasHandle, int n, Complex alpha, [In, Out] Complex[] x); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern float s_dot_product(IntPtr blasHandle, int n, float[] x, float[] y); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern double d_dot_product(IntPtr blasHandle, int n, double[] x, double[] y); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern Complex32 c_dot_product(IntPtr blasHandle, int n, Complex32[] x, Complex32[] y); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern Complex z_dot_product(IntPtr blasHandle, int n, Complex[] x, Complex[] y); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern void s_matrix_multiply(IntPtr blasHandle, int transA, int transB, int m, int n, int k, float alpha, float[] x, float[] y, float beta, [In, Out] float[] c); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern void d_matrix_multiply(IntPtr blasHandle, int transA, int transB, int m, int n, int k, double alpha, double[] x, double[] y, double beta, [In, Out] double[] c); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern void c_matrix_multiply(IntPtr blasHandle, int transA, int transB, int m, int n, int k, Complex32 alpha, Complex32[] x, Complex32[] y, Complex32 beta, [In, Out] Complex32[] c); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern void z_matrix_multiply(IntPtr blasHandle, int transA, int transB, int m, int n, int k, Complex alpha, Complex[] x, Complex[] y, Complex beta, [In, Out] Complex[] c); + + internal static int ToCUDA(this Transpose transpose) + { + switch (transpose) + { + case Transpose.DontTranspose: + return 0; + + case Transpose.Transpose: + return 1; + + case Transpose.ConjugateTranspose: + return 2; + + default: + throw new ArgumentException("Unsupported transpose: " + transpose); + } + } + + #endregion BLAS + + #region LAPACK + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern float s_matrix_norm(byte norm, int rows, int columns, [In] float[] a, [In, Out] float[] work); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern double d_matrix_norm(byte norm, int rows, int columns, [In] double[] a, [In, Out] double[] work); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern float c_matrix_norm(byte norm, int rows, int columns, [In] Complex32[] a, [In, Out] float[] work); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern double z_matrix_norm(byte norm, int rows, int columns, [In] Complex[] a, [In, Out] double[] work); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int s_cholesky_factor(IntPtr solverHandle, int n, [In, Out] float[] a); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int d_cholesky_factor(IntPtr solverHandle, int n, [In, Out] double[] a); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int c_cholesky_factor(IntPtr solverHandle, int n, [In, Out] Complex32[] a); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int z_cholesky_factor(IntPtr solverHandle, int n, [In, Out] Complex[] a); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int s_lu_factor(IntPtr solverHandle, int n, [In, Out] float[] a, [In, Out] int[] ipiv); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int d_lu_factor(IntPtr solverHandle, int n, [In, Out] double[] a, [In, Out] int[] ipiv); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int c_lu_factor(IntPtr solverHandle, int n, [In, Out] Complex32[] a, [In, Out] int[] ipiv); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int z_lu_factor(IntPtr solverHandle, int n, [In, Out] Complex[] a, [In, Out] int[] ipiv); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int s_lu_inverse(IntPtr solverHandle, IntPtr blasHandle, int n, [In, Out] float[] a); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int d_lu_inverse(IntPtr solverHandle, IntPtr blasHandle, int n, [In, Out] double[] a); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int c_lu_inverse(IntPtr solverHandle, IntPtr blasHandle, int n, [In, Out] Complex32[] a); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int z_lu_inverse(IntPtr solverHandle, IntPtr blasHandle, int n, [In, Out] Complex[] a); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int s_lu_inverse_factored(IntPtr blasHandle, int n, [In, Out] float[] a, [In, Out] int[] ipiv); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int d_lu_inverse_factored(IntPtr blasHandle, int n, [In, Out] double[] a, [In, Out] int[] ipiv); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int c_lu_inverse_factored(IntPtr blasHandle, int n, [In, Out] Complex32[] a, [In, Out] int[] ipiv); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int z_lu_inverse_factored(IntPtr blasHandle, int n, [In, Out] Complex[] a, [In, Out] int[] ipiv); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int s_lu_solve_factored(IntPtr solverHandle, int n, int nrhs, float[] a, [In, Out] int[] ipiv, [In, Out] float[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int d_lu_solve_factored(IntPtr solverHandle, int n, int nrhs, double[] a, [In, Out] int[] ipiv, [In, Out] double[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int c_lu_solve_factored(IntPtr solverHandle, int n, int nrhs, Complex32[] a, [In, Out] int[] ipiv, [In, Out] Complex32[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int z_lu_solve_factored(IntPtr solverHandle, int n, int nrhs, Complex[] a, [In, Out] int[] ipiv, [In, Out] Complex[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int s_lu_solve(IntPtr solverHandle, int n, int nrhs, float[] a, [In, Out] float[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int d_lu_solve(IntPtr solverHandle, int n, int nrhs, double[] a, [In, Out] double[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int c_lu_solve(IntPtr solverHandle, int n, int nrhs, Complex32[] a, [In, Out] Complex32[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int z_lu_solve(IntPtr solverHandle, int n, int nrhs, Complex[] a, [In, Out] Complex[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int s_cholesky_solve(IntPtr solverHandle, int n, int nrhs, float[] a, [In, Out] float[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int d_cholesky_solve(IntPtr solverHandle, int n, int nrhs, double[] a, [In, Out] double[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int c_cholesky_solve(IntPtr solverHandle, int n, int nrhs, Complex32[] a, [In, Out] Complex32[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int z_cholesky_solve(IntPtr solverHandle, int n, int nrhs, Complex[] a, [In, Out] Complex[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int s_cholesky_solve_factored(IntPtr solverHandle, int n, int nrhs, float[] a, [In, Out] float[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int d_cholesky_solve_factored(IntPtr solverHandle, int n, int nrhs, double[] a, [In, Out] double[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int c_cholesky_solve_factored(IntPtr solverHandle, int n, int nrhs, Complex32[] a, [In, Out] Complex32[] b); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int z_cholesky_solve_factored(IntPtr solverHandle, int n, int nrhs, Complex[] a, [In, Out] Complex[] b); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int s_qr_factor(int m, int n, [In, Out] float[] r, [In, Out] float[] tau, [In, Out] float[] q, [In, Out] float[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int d_qr_factor(int m, int n, [In, Out] double[] r, [In, Out] double[] tau, [In, Out] double[] q, [In, Out] double[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int c_qr_factor(int m, int n, [In, Out] Complex32[] r, [In, Out] Complex32[] tau, [In, Out] Complex32[] q, [In, Out] Complex32[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int z_qr_factor(int m, int n, [In, Out] Complex[] r, [In, Out] Complex[] tau, [In, Out] Complex[] q, [In, Out] Complex[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int s_qr_thin_factor(int m, int n, [In, Out] float[] q, [In, Out] float[] tau, [In, Out] float[] r, [In, Out] float[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int d_qr_thin_factor(int m, int n, [In, Out] double[] q, [In, Out] double[] tau, [In, Out] double[] r, [In, Out] double[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int c_qr_thin_factor(int m, int n, [In, Out] Complex32[] q, [In, Out] Complex32[] tau, [In, Out] Complex32[] r, [In, Out] Complex32[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int z_qr_thin_factor(int m, int n, [In, Out] Complex[] q, [In, Out] Complex[] tau, [In, Out] Complex[] r, [In, Out] Complex[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int s_qr_solve(int m, int n, int bn, float[] r, float[] b, [In, Out] float[] x, [In, Out] float[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int d_qr_solve(int m, int n, int bn, double[] r, double[] b, [In, Out] double[] x, [In, Out] double[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int c_qr_solve(int m, int n, int bn, Complex32[] r, Complex32[] b, [In, Out] Complex32[] x, [In, Out] Complex32[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int z_qr_solve(int m, int n, int bn, Complex[] r, Complex[] b, [In, Out] Complex[] x, [In, Out] Complex[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int s_qr_solve_factored(int m, int n, int bn, float[] r, float[] b, float[] tau, [In, Out] float[] x, [In, Out] float[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int d_qr_solve_factored(int m, int n, int bn, double[] r, double[] b, double[] tau, [In, Out] double[] x, [In, Out] double[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int c_qr_solve_factored(int m, int n, int bn, Complex32[] r, Complex32[] b, Complex32[] tau, [In, Out] Complex32[] x, [In, Out] Complex32[] work, int len); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int z_qr_solve_factored(int m, int n, int bn, Complex[] r, Complex[] b, Complex[] tau, [In, Out] Complex[] x, [In, Out] Complex[] work, int len); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int s_svd_factor(IntPtr solverHandle, [MarshalAs(UnmanagedType.U1)] bool computeVectors, int m, int n, [In, Out] float[] a, [In, Out] float[] s, [In, Out] float[] u, [In, Out] float[] v); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int d_svd_factor(IntPtr solverHandle, [MarshalAs(UnmanagedType.U1)] bool computeVectors, int m, int n, [In, Out] double[] a, [In, Out] double[] s, [In, Out] double[] u, [In, Out] double[] v); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int c_svd_factor(IntPtr solverHandle, [MarshalAs(UnmanagedType.U1)] bool computeVectors, int m, int n, [In, Out] Complex32[] a, [In, Out] Complex32[] s, [In, Out] Complex32[] u, [In, Out] Complex32[] v); + + [DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + internal static extern int z_svd_factor(IntPtr solverHandle, [MarshalAs(UnmanagedType.U1)] bool computeVectors, int m, int n, [In, Out] Complex[] a, [In, Out] Complex[] s, [In, Out] Complex[] u, [In, Out] Complex[] v); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int s_eigen([MarshalAs(UnmanagedType.U1)] bool isSymmetric, int n, [In] float[] a, [In, Out] float[] vectors, [In, Out] Complex[] values, [In, Out] float[] d); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int d_eigen([MarshalAs(UnmanagedType.U1)] bool isSymmetric, int n, [In] double[] a, [In, Out] double[] vectors, [In, Out] Complex[] values, [In, Out] double[] d); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int c_eigen([MarshalAs(UnmanagedType.U1)] bool isSymmetric, int n, [In] Complex32[] a, [In, Out] Complex32[] vectors, [In, Out] Complex[] values, [In, Out] Complex32[] d); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern int z_eigen([MarshalAs(UnmanagedType.U1)] bool isSymmetric, int n, [In] Complex[] a, [In, Out] Complex[] vectors, [In, Out] Complex[] values, [In, Out] Complex[] d); + + #endregion LAPACK + + #region Vector Functions + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void s_vector_add(int n, float[] x, float[] y, [In, Out] float[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void s_vector_subtract(int n, float[] x, float[] y, [In, Out] float[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void s_vector_multiply(int n, float[] x, float[] y, [In, Out] float[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void s_vector_divide(int n, float[] x, float[] y, [In, Out] float[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void d_vector_add(int n, double[] x, double[] y, [In, Out] double[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void d_vector_subtract(int n, double[] x, double[] y, [In, Out] double[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void d_vector_multiply(int n, double[] x, double[] y, [In, Out] double[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void d_vector_divide(int n, double[] x, double[] y, [In, Out] double[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void c_vector_add(int n, Complex32[] x, Complex32[] y, [In, Out] Complex32[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void c_vector_subtract(int n, Complex32[] x, Complex32[] y, [In, Out] Complex32[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void c_vector_multiply(int n, Complex32[] x, Complex32[] y, [In, Out] Complex32[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void c_vector_divide(int n, Complex32[] x, Complex32[] y, [In, Out] Complex32[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void z_vector_add(int n, Complex[] x, Complex[] y, [In, Out] Complex[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void z_vector_subtract(int n, Complex[] x, Complex[] y, [In, Out] Complex[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void z_vector_multiply(int n, Complex[] x, Complex[] y, [In, Out] Complex[] result); + + //[DllImport(_DllName, ExactSpelling = true, SetLastError = false, CallingConvention = CallingConvention.Cdecl)] + //internal static extern void z_vector_divide(int n, Complex[] x, Complex[] y, [In, Out] Complex[] result); + + #endregion Vector Functions + + // ReSharper restore InconsistentNaming + } +} + +#endif diff --git a/src/UnitTests/UnitTests-CUDA.csproj b/src/UnitTests/UnitTests-CUDA.csproj new file mode 100644 index 00000000..598ce9a2 --- /dev/null +++ b/src/UnitTests/UnitTests-CUDA.csproj @@ -0,0 +1,348 @@ + + + + 10.0 + Debug + AnyCPU + 8.0.30703 + 2.0 + {E79C0395-01DC-4BC9-B86C-ED45790892C5} + Library + Properties + MathNet.Numerics.UnitTests + MathNet.Numerics.UnitTestsCUDA + v4.5 + 512 + ..\..\ + + + TRACE;CUDA + ..\..\out\CUDA\Windows\ + ..\..\obj\CUDA\Windows\x86\ + ..\..\obj\CUDA\Windows\x86\ + true + pdbonly + prompt + MinimumRecommendedRules.ruleset + 1591 + AnyCPU + + + TRACE;DEBUG;CUDA + ..\..\out\CUDA\Windows\ + ..\..\obj\CUDA\Windows\x86\ + ..\..\obj\CUDA\Windows\x86\ + false + full + true + prompt + 4 + 1591 + AnyCPU + + + + + + + + + + + + + + + + + + + data\Codeplex-5667.csv + Always + + + data\Github-Cureos-1.csv + Always + + + data\Matlab\A.mat + Always + + + data\Matlab\collection-nocompress.mat + Always + + + data\Matlab\collection.mat + Always + + + data\Matlab\complex.mat + Always + + + data\Matlab\sparse-large.mat + Always + + + data\Matlab\sparse-small.mat + Always + + + data\Matlab\sparse_complex.mat + Always + + + data\Matlab\v.mat + Always + + + data\NIST\AtmWtAgt.dat + Always + + + data\NIST\Bennett5.dat + Always + + + data\NIST\BoxBOD.dat + Always + + + data\NIST\Chwirut1.dat + Always + + + data\NIST\Chwirut2.dat + Always + + + data\NIST\DanWood.dat + Always + + + data\NIST\Eckerle4.dat + Always + + + data\NIST\ENSO.dat + Always + + + data\NIST\Filip.dat + Always + + + data\NIST\Gauss1.dat + Always + + + data\NIST\Gauss2.dat + Always + + + data\NIST\Gauss3.dat + Always + + + data\NIST\Hahn1.dat + Always + + + data\NIST\Kirby2.dat + Always + + + data\NIST\Lanczos1.dat + Always + + + data\NIST\Lanczos2.dat + Always + + + data\NIST\Lanczos3.dat + Always + + + data\NIST\Lew.dat + Always + + + data\NIST\Longley.dat + Always + + + data\NIST\Lottery.dat + Always + + + data\NIST\Mavro.dat + Always + + + data\NIST\MGH09.dat + Always + + + data\NIST\MGH10.dat + Always + + + data\NIST\MGH17.dat + Always + + + data\NIST\Michelso.dat + Always + + + data\NIST\Misra1a.dat + Always + + + data\NIST\Misra1b.dat + Always + + + data\NIST\Misra1c.dat + Always + + + data\NIST\Misra1d.dat + Always + + + data\NIST\Nelson.dat + Always + + + data\NIST\NoInt1.dat + Always + + + data\NIST\NoInt2.dat + Always + + + data\NIST\Norris.dat + Always + + + data\NIST\NumAcc1.dat + Always + + + data\NIST\NumAcc2.dat + Always + + + data\NIST\NumAcc3.dat + Always + + + data\NIST\NumAcc4.dat + Always + + + data\NIST\Pontius.dat + Always + + + data\NIST\Rat42.dat + Always + + + data\NIST\Rat43.dat + Always + + + data\NIST\Roszman1.dat + Always + + + data\NIST\SiRstvt.dat + Always + + + data\NIST\SmLs01t.dat + Always + + + data\NIST\SmLs02t.dat + Always + + + data\NIST\SmLs03t.dat + Always + + + data\NIST\SmLs04t.dat + Always + + + data\NIST\SmLs05t.dat + Always + + + data\NIST\SmLs06t.dat + Always + + + data\NIST\SmLs07t.dat + Always + + + data\NIST\SmLs08t.dat + Always + + + data\NIST\SmLs09t.dat + Always + + + data\NIST\Thurber.dat + Always + + + data\NIST\Wampler1.dat + Always + + + data\NIST\Wampler2.dat + Always + + + data\NIST\Wampler3.dat + Always + + + data\NIST\Wampler4.dat + Always + + + data\NIST\Wampler5.dat + Always + + + + data\NIST\Meixner.dat + Always + + + + + + {b7cae5f4-a23f-4438-b5be-41226618b695} + Numerics + + + + + + ..\..\packages\NUnit\lib\nunit.framework.dll + True + True + + + \ No newline at end of file diff --git a/src/UnitTests/UseLinearAlgebraProvider.cs b/src/UnitTests/UseLinearAlgebraProvider.cs index fc4a42d6..d45957e8 100644 --- a/src/UnitTests/UseLinearAlgebraProvider.cs +++ b/src/UnitTests/UseLinearAlgebraProvider.cs @@ -40,6 +40,9 @@ namespace MathNet.Numerics.UnitTests { #if !NET35 && NATIVE Control.UseNativeMKL(); +#endif +#if CUDA + Control.UseNativeCUDA(); #endif } From 547745dd22e4546ab6e8f93a5222fe8354ea92ec Mon Sep 17 00:00:00 2001 From: Matthew Johnson Date: Sat, 25 Apr 2015 18:45:12 +0100 Subject: [PATCH 05/14] Fix for the matrix inverse bug. --- src/NativeProviders/CUDA/lapack.cpp | 79 ++++++++++++++--------------- 1 file changed, 37 insertions(+), 42 deletions(-) diff --git a/src/NativeProviders/CUDA/lapack.cpp b/src/NativeProviders/CUDA/lapack.cpp index 718c4909..d0a7c96d 100644 --- a/src/NativeProviders/CUDA/lapack.cpp +++ b/src/NativeProviders/CUDA/lapack.cpp @@ -43,8 +43,8 @@ inline int lu_factor(cusolverDnHandle_t solverHandle, int m, T a[], int ipiv[], return info; }; -template -inline int lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int n, T a[], GETRF getrf, GETRI getri, GETRFBSIZE getrfbsize) +template +inline int lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int n, T a[], GETRF getrf, GETRIBATCHED getribatched, GETRFBSIZE getrfbsize) { int info = 0; @@ -63,13 +63,8 @@ inline int lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle int* d_info = NULL; cudaMalloc((void**)&d_info, sizeof(int)); - printf("initial %f %f %f %f %f %f %f %f %f\r\n", a[0], a[1], a[2], a[3], a[4], a[5], a[6], a[7], a[8]); - getrf(solverHandle, n, n, d_A, n, work, d_I, d_info); cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); - - cublasGetMatrix(n, n, sizeof(T), d_A, n, a, n); - printf("after factor %f %f %f %f %f %f %f %f %f\r\n", a[0], a[1], a[2], a[3], a[4], a[5], a[6], a[7], a[8]); cudaFree(work); @@ -83,20 +78,26 @@ inline int lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle T* d_C = NULL; cudaMalloc((void**)&d_C, n*n*sizeof(T)); - - getri(blasHandle, n, d_A, n, d_I, d_C, n, d_info); - cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); - cublasGetMatrix(n, n, sizeof(T), d_A, n, a, n); - printf("a inverse %f %f %f %f %f %f %f %f %f\r\n", a[0], a[1], a[2], a[3], a[4], a[5], a[6], a[7], a[8]); + const T **d_Aarray = NULL; + cudaMalloc((void**)&d_Aarray, sizeof(T*)); + cudaMemcpy(d_Aarray, &d_A, sizeof(T*), cudaMemcpyHostToDevice); + + T **d_Carray = NULL; + cudaMalloc((void**)&d_Carray, sizeof(T*)); + cudaMemcpy(d_Carray, &d_C, sizeof(T*), cudaMemcpyHostToDevice); + + getribatched(blasHandle, n, d_Aarray, n, d_I, d_Carray, n, d_info, 1); + cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); cublasGetMatrix(n, n, sizeof(T), d_C, n, a, n); - printf("c inverse %f %f %f %f %f %f %f %f %f\r\n", a[0], a[1], a[2], a[3], a[4], a[5], a[6], a[7], a[8]); cudaFree(d_A); cudaFree(d_I); cudaFree(d_C); cudaFree(d_info); + cudaFree(d_Aarray); + cudaFree(d_Carray); return info; }; @@ -122,7 +123,15 @@ inline int lu_inverse_factored(cublasHandle_t blasHandle, int n, T a[], int ipiv int* d_info = NULL; cudaMalloc((void**)&d_info, sizeof(int)); - getri(blasHandle, n, d_A, n, d_I, d_C, n, d_info); + const T **d_Aarray = NULL; + cudaMalloc((void**)&d_Aarray, sizeof(T*)); + cudaMemcpy(d_Aarray, &d_A, sizeof(T*), cudaMemcpyHostToDevice); + + T **d_Carray = NULL; + cudaMalloc((void**)&d_Carray, sizeof(T*)); + cudaMemcpy(d_Carray, &d_C, sizeof(T*), cudaMemcpyHostToDevice); + + getri(blasHandle, n, d_Aarray, n, d_I, d_Carray, n, d_info, 1); cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); cublasGetMatrix(n, n, sizeof(T), d_C, n, a, n); @@ -134,6 +143,8 @@ inline int lu_inverse_factored(cublasHandle_t blasHandle, int n, T a[], int ipiv cudaFree(d_I); cudaFree(d_C); cudaFree(d_info); + cudaFree(d_Aarray); + cudaFree(d_Carray); return info; } @@ -711,26 +722,10 @@ inline int complex_svd_factor(cusolverDnHandle_t solverHandle, bool compute_vect #define cgesvdbsize cusolverDnCgesvd_bufferSize #define zgesvdbsize cusolverDnZgesvd_bufferSize - -inline int sgetri(cublasHandle_t handle, int n, const float a[], int lda, const int ipiv[], float c[], int ldc, int *info) -{ - return cublasSgetriBatched(handle, n, &a, lda, ipiv, &c, ldc, info, 1); -} - -inline int dgetri(cublasHandle_t handle, int n, const double a[], int lda, const int ipiv[], double c[], int ldc, int *info) -{ - return cublasDgetriBatched(handle, n, &a, lda, ipiv, &c, ldc, info, 1); -} - -inline int cgetri(cublasHandle_t handle, int n, const cuComplex a[], int lda, const int ipiv[], cuComplex c[], int ldc, int *info) -{ - return cublasCgetriBatched(handle, n, &a, lda, ipiv, &c, ldc, info, 1); -} - -inline int zgetri(cublasHandle_t handle, int n, const cuDoubleComplex a[], int lda, const int ipiv[], cuDoubleComplex c[], int ldc, int *info) -{ - return cublasZgetriBatched(handle, n, &a, lda, ipiv, &c, ldc, info, 1); -} +#define sgetribatched cublasSgetriBatched +#define dgetribatched cublasDgetriBatched +#define cgetribatched cublasCgetriBatched +#define zgetribatched cublasZgetriBatched extern "C" { @@ -756,42 +751,42 @@ extern "C" { DLLEXPORT int s_lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int n, float a[]) { - return lu_inverse(solverHandle, blasHandle, n, a, sgetrf, sgetri, sgetrfbsize); + return lu_inverse(solverHandle, blasHandle, n, a, sgetrf, sgetribatched, sgetrfbsize); } DLLEXPORT int d_lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int n, double a[]) { - return lu_inverse(solverHandle, blasHandle, n, a, dgetrf, dgetri, dgetrfbsize); + return lu_inverse(solverHandle, blasHandle, n, a, dgetrf, dgetribatched, dgetrfbsize); } DLLEXPORT int c_lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int n, cuComplex a[]) { - return lu_inverse(solverHandle, blasHandle, n, a, cgetrf, cgetri, cgetrfbsize); + return lu_inverse(solverHandle, blasHandle, n, a, cgetrf, cgetribatched, cgetrfbsize); } DLLEXPORT int z_lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle, int n, cuDoubleComplex a[]) { - return lu_inverse(solverHandle, blasHandle, n, a, zgetrf, zgetri, zgetrfbsize); + return lu_inverse(solverHandle, blasHandle, n, a, zgetrf, zgetribatched, zgetrfbsize); } DLLEXPORT int s_lu_inverse_factored(cublasHandle_t blasHandle, int n, float a[], int ipiv[]) { - return lu_inverse_factored(blasHandle, n, a, ipiv, sgetri); + return lu_inverse_factored(blasHandle, n, a, ipiv, sgetribatched); } DLLEXPORT int d_lu_inverse_factored(cublasHandle_t blasHandle, int n, double a[], int ipiv[]) { - return lu_inverse_factored(blasHandle, n, a, ipiv, dgetri); + return lu_inverse_factored(blasHandle, n, a, ipiv, dgetribatched); } DLLEXPORT int c_lu_inverse_factored(cublasHandle_t blasHandle, int n, cuComplex a[], int ipiv[]) { - return lu_inverse_factored(blasHandle, n, a, ipiv, cgetri); + return lu_inverse_factored(blasHandle, n, a, ipiv, cgetribatched); } DLLEXPORT int z_lu_inverse_factored(cublasHandle_t blasHandle, int n, cuDoubleComplex a[], int ipiv[]) { - return lu_inverse_factored(blasHandle, n, a, ipiv, zgetri); + return lu_inverse_factored(blasHandle, n, a, ipiv, zgetribatched); } DLLEXPORT int s_lu_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, float a[], int ipiv[], float b[]) From 3de6b41527ff82eb5172c72b1de18522b284f064 Mon Sep 17 00:00:00 2001 From: Matthew Johnson Date: Sun, 26 Apr 2015 01:57:21 +0100 Subject: [PATCH 06/14] Fixed some issues, but potrs doesn't seem to be working at all. --- src/NativeProviders/CUDA/lapack.cpp | 32 +++++++++---------- .../Single/LinearAlgebraProviderTests.cs | 20 ++++++++++++ 2 files changed, 36 insertions(+), 16 deletions(-) diff --git a/src/NativeProviders/CUDA/lapack.cpp b/src/NativeProviders/CUDA/lapack.cpp index d0a7c96d..e92ae176 100644 --- a/src/NativeProviders/CUDA/lapack.cpp +++ b/src/NativeProviders/CUDA/lapack.cpp @@ -28,7 +28,7 @@ inline int lu_factor(cusolverDnHandle_t solverHandle, int m, T a[], int ipiv[], getrf(solverHandle, m, m, d_A, m, work, d_I, d_info); - cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); + cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); cublasGetMatrix(m, m, sizeof(T), d_A, m, a, m); cublasGetVector(m, sizeof(T), d_I, 1, ipiv, 1); @@ -64,7 +64,7 @@ inline int lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle cudaMalloc((void**)&d_info, sizeof(int)); getrf(solverHandle, n, n, d_A, n, work, d_I, d_info); - cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); + cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); cudaFree(work); @@ -88,7 +88,7 @@ inline int lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle cudaMemcpy(d_Carray, &d_C, sizeof(T*), cudaMemcpyHostToDevice); getribatched(blasHandle, n, d_Aarray, n, d_I, d_Carray, n, d_info, 1); - cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); + cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); cublasGetMatrix(n, n, sizeof(T), d_C, n, a, n); @@ -132,7 +132,7 @@ inline int lu_inverse_factored(cublasHandle_t blasHandle, int n, T a[], int ipiv cudaMemcpy(d_Carray, &d_C, sizeof(T*), cudaMemcpyHostToDevice); getri(blasHandle, n, d_Aarray, n, d_I, d_Carray, n, d_info, 1); - cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); + cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); cublasGetMatrix(n, n, sizeof(T), d_C, n, a, n); cublasGetVector(n, sizeof(int), d_I, 1, ipiv, 1); @@ -172,7 +172,7 @@ inline int lu_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, T cudaMalloc((void**)&d_info, sizeof(int)); getrs(solverHandle, CUBLAS_OP_N, n, nrhs, d_A, n, d_I, d_B, n, d_info); - cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); + cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); cublasGetMatrix(n, nrhs, sizeof(T), d_B, n, b, n); @@ -207,7 +207,7 @@ inline int lu_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[], T b cudaMalloc((void**)&d_info, sizeof(int)); getrf(solverHandle, n, n, d_A, n, work, d_I, d_info); - cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); + cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); if (info != 0) { @@ -236,7 +236,7 @@ inline int lu_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[], T b template -inline int cholesky_factor(cusolverDnHandle_t solverHandle, int n, T* a, POTRF potrf, POTRFBSIZE potrfbsize) +inline int cholesky_factor(cusolverDnHandle_t solverHandle, int n, T a[], POTRF potrf, POTRFBSIZE potrfbsize) { int info = 0; @@ -253,7 +253,7 @@ inline int cholesky_factor(cusolverDnHandle_t solverHandle, int n, T* a, POTRF p cudaMalloc((void**)&d_info, sizeof(int)); potrf(solverHandle, CUBLAS_FILL_MODE_LOWER, n, d_A, n, work, lWork, d_info); - cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); + cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); cublasGetMatrix(n, n, sizeof(T), d_A, n, a, n); @@ -279,7 +279,7 @@ inline int cholesky_factor(cusolverDnHandle_t solverHandle, int n, T* a, POTRF p template inline int cholesky_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[], T b[], POTRF potrf, POTRS potrs, POTRFBSIZE potrfbsize) { - int info; + int info = 0; T* d_A = NULL; cudaMalloc((void**)&d_A, n*n*sizeof(T)); @@ -294,7 +294,7 @@ inline int cholesky_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[ cudaMalloc((void**)&d_info, sizeof(int)); potrf(solverHandle, CUBLAS_FILL_MODE_LOWER, n, d_A, n, work, lWork, d_info); - cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); + cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); cudaFree(work); @@ -310,7 +310,7 @@ inline int cholesky_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[ cublasSetMatrix(n, nrhs, sizeof(T), b, n, d_B, n); potrs(solverHandle, CUBLAS_FILL_MODE_LOWER, n, nrhs, d_A, n, d_B, n, d_info); - cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); + cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); cublasGetMatrix(n, nrhs, sizeof(T), d_B, n, b, n); @@ -324,7 +324,7 @@ inline int cholesky_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[ template inline int cholesky_solve_factored(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[], T b[], POTRS potrs) { - int info; + int info = 0; T* d_A = NULL; cudaMalloc((void**)&d_A, n*n*sizeof(T)); @@ -338,7 +338,7 @@ inline int cholesky_solve_factored(cusolverDnHandle_t solverHandle, int n, int n cudaMalloc((void**)&d_info, sizeof(int)); potrs(solverHandle, CUBLAS_FILL_MODE_LOWER, n, nrhs, d_A, n, d_B, n, d_info); - cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); + cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); cublasGetMatrix(n, nrhs, sizeof(T), d_B, n, b, n); @@ -477,7 +477,7 @@ inline int svd_factor(cusolverDnHandle_t solverHandle, bool compute_vectors, int char job = compute_vectors ? 'A' : 'N'; gesvd(solverHandle, job, job, m, n, d_A, m, d_S, d_U, m, d_V, n, work, lWork, rwork, d_info); - cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); + cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); cublasGetVector(dim_s, sizeof(T), d_S, 1, s, 1); cublasGetMatrix(m, m, sizeof(T), d_U, m, u, m); @@ -527,7 +527,7 @@ inline int complex_svd_factor(cusolverDnHandle_t solverHandle, bool compute_vect char job = compute_vectors ? 'A' : 'N'; gesvd(solverHandle, job, job, m, n, d_A, m, d_S, d_U, m, d_V, n, work, lWork, rwork, d_info); - cudaMemcpy(&info, d_info, 1, cudaMemcpyDeviceToHost); + cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); cublasGetVector(dim_s, sizeof(T), d_S, 1, s_local, 1); cublasGetMatrix(m, m, sizeof(T), d_U, m, u, m); @@ -834,7 +834,7 @@ extern "C" { return cholesky_factor(solverHandle, n, a, spotrf, spotrfbsize); } - DLLEXPORT int d_cholesky_factor(cusolverDnHandle_t solverHandle, int n, double* a) + DLLEXPORT int d_cholesky_factor(cusolverDnHandle_t solverHandle, int n, double a[]) { return cholesky_factor(solverHandle, n, a, dpotrf, dpotrfbsize); } diff --git a/src/UnitTests/LinearAlgebraProviderTests/Single/LinearAlgebraProviderTests.cs b/src/UnitTests/LinearAlgebraProviderTests/Single/LinearAlgebraProviderTests.cs index 1ed7b3f0..fab970db 100644 --- a/src/UnitTests/LinearAlgebraProviderTests/Single/LinearAlgebraProviderTests.cs +++ b/src/UnitTests/LinearAlgebraProviderTests/Single/LinearAlgebraProviderTests.cs @@ -446,7 +446,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Single var a = new float[matrix.RowCount*matrix.RowCount]; Array.Copy(matrix.Values, a, a.Length); +#if CUDA + float[] work = null; +#else var work = new float[matrix.RowCount]; +#endif Control.LinearAlgebraProvider.LUInverse(a, matrix.RowCount, work); AssertHelpers.AlmostEqual(a[0], -0.454545454545454, 5); @@ -475,7 +479,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Single Control.LinearAlgebraProvider.LUFactor(a, matrix.RowCount, ipiv); +#if CUDA + float[] work = null; +#else var work = new float[matrix.RowCount]; +#endif Control.LinearAlgebraProvider.LUInverseFactored(a, matrix.RowCount, ipiv, work); AssertHelpers.AlmostEqual(a[0], -0.454545454545454, 5); @@ -1450,7 +1458,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Single var s = new float[matrix.RowCount]; var u = new float[matrix.RowCount*matrix.RowCount]; var vt = new float[matrix.ColumnCount*matrix.ColumnCount]; +#if CUDA + float[] work = null; +#else var work = new float[100]; +#endif Control.LinearAlgebraProvider.SingularValueDecomposition(true, a, matrix.RowCount, matrix.ColumnCount, s, u, vt, work); @@ -1489,7 +1501,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Single var s = new float[matrix.ColumnCount]; var u = new float[matrix.RowCount*matrix.RowCount]; var vt = new float[matrix.ColumnCount*matrix.ColumnCount]; +#if CUDA + float[] work = null; +#else var work = new float[100]; +#endif Control.LinearAlgebraProvider.SingularValueDecomposition(true, a, matrix.RowCount, matrix.ColumnCount, s, u, vt, work); @@ -1525,7 +1541,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Single var s = new float[matrix.RowCount]; var u = new float[matrix.RowCount*matrix.RowCount]; var vt = new float[matrix.ColumnCount*matrix.ColumnCount]; +#if CUDA + float[] work = null; +#else var work = new float[100]; +#endif Control.LinearAlgebraProvider.SingularValueDecomposition(true, a, matrix.RowCount, matrix.ColumnCount, s, u, vt, work); From bc4b92655768ffdc409df49ad69313350a33f577 Mon Sep 17 00:00:00 2001 From: Matthew Johnson Date: Sun, 26 Apr 2015 02:24:05 +0100 Subject: [PATCH 07/14] Fixed a bug with GEMM --- src/NativeProviders/CUDA/blas.cpp | 25 ++++++++++++++----------- 1 file changed, 14 insertions(+), 11 deletions(-) diff --git a/src/NativeProviders/CUDA/blas.cpp b/src/NativeProviders/CUDA/blas.cpp index c8348a32..fbad097e 100644 --- a/src/NativeProviders/CUDA/blas.cpp +++ b/src/NativeProviders/CUDA/blas.cpp @@ -1,3 +1,4 @@ +#include #include "cublas_v2.h" #include "cuda_runtime.h" #include "wrapper_common.h" @@ -54,19 +55,21 @@ void cuda_dot(const cublasHandle_t blasHandle, const int n, const T x[], int inc } template -void cuda_gemm(const cublasHandle_t handle, const cublasOperation_t transa, const cublasOperation_t transb, int m, int n, int k, const T *alpha, const T A[], int lda, const T B[], int ldb, const T *beta, T C[], int ldc, GEMM gemm) +void cuda_gemm(const cublasHandle_t handle, const cublasOperation_t transa, const cublasOperation_t transb, int m, int n, int k, const T alpha, const T A[], int lda, const T B[], int ldb, const T beta, T C[], int ldc, GEMM gemm) { T *d_A = NULL; - T *d_B = NULL; - T *d_C = NULL; cudaMalloc((void**)&d_A, m*k*sizeof(T)); - cudaMalloc((void**)&d_B, k*n*sizeof(T)); - cudaMalloc((void**)&d_C, m*n*sizeof(T)); - cublasSetMatrix(m, k, sizeof(T), A, m, d_A, m); + + T *d_B = NULL; + cudaMalloc((void**)&d_B, k*n*sizeof(T)); cublasSetMatrix(k, n, sizeof(T), B, k, d_B, k); - gemm(handle, transa, transb, m, n, k, alpha, d_A, lda, d_B, ldb, beta, d_C, ldc); + T *d_C = NULL; + cudaMalloc((void**)&d_C, m*n*sizeof(T)); + cublasSetMatrix(m, n, sizeof(T), C, m, d_C, m); + + gemm(handle, transa, transb, m, n, k, &alpha, d_A, lda, d_B, ldb, &beta, d_C, ldc); cublasGetMatrix(m, n, sizeof(T), d_C, m, C, m); @@ -137,28 +140,28 @@ extern "C" { int lda = transA == CUBLAS_OP_N ? m : k; int ldb = transB == CUBLAS_OP_N ? k : n; - cuda_gemm(blasHandle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m, cublasSgemm); + cuda_gemm(blasHandle, transA, transB, m, n, k, alpha, x, lda, y, ldb, beta, c, m, cublasSgemm); } DLLEXPORT void d_matrix_multiply(const cublasHandle_t blasHandle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const double alpha, const double x[], const double y[], const double beta, double c[]){ int lda = transA == CUBLAS_OP_N ? m : k; int ldb = transB == CUBLAS_OP_N ? k : n; - cuda_gemm(blasHandle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m, cublasDgemm); + cuda_gemm(blasHandle, transA, transB, m, n, k, alpha, x, lda, y, ldb, beta, c, m, cublasDgemm); } DLLEXPORT void c_matrix_multiply(const cublasHandle_t blasHandle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const cuComplex alpha, const cuComplex x[], const cuComplex y[], const cuComplex beta, cuComplex c[]){ int lda = transA == CUBLAS_OP_N ? m : k; int ldb = transB == CUBLAS_OP_N ? k : n; - cuda_gemm(blasHandle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m, cublasCgemm); + cuda_gemm(blasHandle, transA, transB, m, n, k, alpha, x, lda, y, ldb, beta, c, m, cublasCgemm); } DLLEXPORT void z_matrix_multiply(const cublasHandle_t blasHandle, cublasOperation_t transA, cublasOperation_t transB, const int m, const int n, const int k, const cuDoubleComplex alpha, const cuDoubleComplex x[], const cuDoubleComplex y[], const cuDoubleComplex beta, cuDoubleComplex c[]){ int lda = transA == CUBLAS_OP_N ? m : k; int ldb = transB == CUBLAS_OP_N ? k : n; - cuda_gemm(blasHandle, transA, transB, m, n, k, &alpha, x, lda, y, ldb, &beta, c, m, cublasZgemm); + cuda_gemm(blasHandle, transA, transB, m, n, k, alpha, x, lda, y, ldb, beta, c, m, cublasZgemm); } } From 8d06f9d4354b31a2a1f2b49c55472469c2319e36 Mon Sep 17 00:00:00 2001 From: Matthew Johnson Date: Sun, 26 Apr 2015 19:22:17 +0100 Subject: [PATCH 08/14] Passing most tests now, investigating the stragglers. --- src/NativeProviders/CUDA/blas.cpp | 24 +++++++++---------- src/NativeProviders/CUDA/lapack.cpp | 15 ++++++------ .../Cuda/CudaLinearAlgebraProvider.Complex.cs | 10 ++++---- .../CudaLinearAlgebraProvider.Complex32.cs | 12 ++++++---- .../Cuda/CudaLinearAlgebraProvider.Double.cs | 13 +++++----- .../Cuda/CudaLinearAlgebraProvider.Single.cs | 8 ++++--- .../Cuda/CudaLinearAlgebraProvider.cs | 4 ++-- .../Complex/LinearAlgebraProviderTests.cs | 20 ++++++++++++++++ .../Complex32/LinearAlgebraProviderTests.cs | 20 ++++++++++++++++ .../Double/LinearAlgebraProviderTests.cs | 20 ++++++++++++++++ 10 files changed, 106 insertions(+), 40 deletions(-) diff --git a/src/NativeProviders/CUDA/blas.cpp b/src/NativeProviders/CUDA/blas.cpp index fbad097e..b99559b6 100644 --- a/src/NativeProviders/CUDA/blas.cpp +++ b/src/NativeProviders/CUDA/blas.cpp @@ -4,7 +4,7 @@ #include "wrapper_common.h" template -void cuda_axpy(const cublasHandle_t blasHandle, const int n, const T *alpha, const T x[], int incX, T y[], int incY, AXPY axpy) +void cuda_axpy(const cublasHandle_t blasHandle, const int n, const T alpha, const T x[], int incX, T y[], int incY, AXPY axpy) { T *d_X = NULL; T *d_Y = NULL; @@ -14,7 +14,7 @@ void cuda_axpy(const cublasHandle_t blasHandle, const int n, const T *alpha, con cublasSetVector(n, sizeof(T), x, incX, d_X, incX); cublasSetVector(n, sizeof(T), y, incY, d_Y, incY); - axpy(blasHandle, n, alpha, d_X, incX, d_Y, incX); + axpy(blasHandle, n, &alpha, d_X, incX, d_Y, incX); cublasGetVector(n, sizeof(T), d_Y, incY, y, incY); @@ -23,14 +23,14 @@ void cuda_axpy(const cublasHandle_t blasHandle, const int n, const T *alpha, con } template -void cuda_scal(const cublasHandle_t blasHandle, const int n, const T *alpha, T x[], int incX, SCAL scal) +void cuda_scal(const cublasHandle_t blasHandle, const int n, const T alpha, T x[], int incX, SCAL scal) { T *d_X = NULL; cudaMalloc((void**)&d_X, n*sizeof(T)); cublasSetVector(n, sizeof(T), x, incX, d_X, incX); - scal(blasHandle, n, alpha, d_X, incX); + scal(blasHandle, n, &alpha, d_X, incX); cublasGetVector(n, sizeof(T), d_X, incX, x, incX); @@ -81,35 +81,35 @@ void cuda_gemm(const cublasHandle_t handle, const cublasOperation_t transa, cons extern "C" { DLLEXPORT void s_axpy(const cublasHandle_t blasHandle, const int n, const float alpha, const float x[], float y[]){ - cuda_axpy(blasHandle, n, &alpha, x, 1, y, 1, cublasSaxpy); + cuda_axpy(blasHandle, n, alpha, x, 1, y, 1, cublasSaxpy); } DLLEXPORT void d_axpy(const cublasHandle_t blasHandle, const int n, const double alpha, const double x[], double y[]){ - cuda_axpy(blasHandle, n, &alpha, x, 1, y, 1, cublasDaxpy); + cuda_axpy(blasHandle, n, alpha, x, 1, y, 1, cublasDaxpy); } DLLEXPORT void c_axpy(const cublasHandle_t blasHandle, const int n, const cuComplex alpha, const cuComplex x[], cuComplex y[]){ - cuda_axpy(blasHandle, n, &alpha, x, 1, y, 1, cublasCaxpy); + cuda_axpy(blasHandle, n, alpha, x, 1, y, 1, cublasCaxpy); } DLLEXPORT void z_axpy(const cublasHandle_t blasHandle, const int n, const cuDoubleComplex alpha, const cuDoubleComplex x[], cuDoubleComplex y[]){ - cuda_axpy(blasHandle, n, &alpha, x, 1, y, 1, cublasZaxpy); + cuda_axpy(blasHandle, n, alpha, x, 1, y, 1, cublasZaxpy); } DLLEXPORT void s_scale(const cublasHandle_t blasHandle, const int n, const float alpha, float x[]){ - cuda_scal(blasHandle, n, &alpha, x, 1, cublasSscal); + cuda_scal(blasHandle, n, alpha, x, 1, cublasSscal); } DLLEXPORT void d_scale(const cublasHandle_t blasHandle, const int n, const double alpha, double x[]){ - cuda_scal(blasHandle, n, &alpha, x, 1, cublasDscal); + cuda_scal(blasHandle, n, alpha, x, 1, cublasDscal); } DLLEXPORT void c_scale(const cublasHandle_t blasHandle, const int n, const cuComplex alpha, cuComplex x[]){ - cuda_scal(blasHandle, n, &alpha, x, 1, cublasCscal); + cuda_scal(blasHandle, n, alpha, x, 1, cublasCscal); } DLLEXPORT void z_scale(const cublasHandle_t blasHandle, const int n, const cuDoubleComplex alpha, cuDoubleComplex x[]){ - cuda_scal(blasHandle, n, &alpha, x, 1, cublasZscal); + cuda_scal(blasHandle, n, alpha, x, 1, cublasZscal); } DLLEXPORT float s_dot_product(const cublasHandle_t blasHandle, const int n, const float x[], const float y[]){ diff --git a/src/NativeProviders/CUDA/lapack.cpp b/src/NativeProviders/CUDA/lapack.cpp index e92ae176..b4cbf1e7 100644 --- a/src/NativeProviders/CUDA/lapack.cpp +++ b/src/NativeProviders/CUDA/lapack.cpp @@ -31,7 +31,7 @@ inline int lu_factor(cusolverDnHandle_t solverHandle, int m, T a[], int ipiv[], cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); cublasGetMatrix(m, m, sizeof(T), d_A, m, a, m); - cublasGetVector(m, sizeof(T), d_I, 1, ipiv, 1); + cublasGetVector(m, sizeof(int), d_I, 1, ipiv, 1); shift_ipiv_down(m, ipiv); @@ -49,7 +49,7 @@ inline int lu_inverse(cusolverDnHandle_t solverHandle, cublasHandle_t blasHandle int info = 0; int* d_I = NULL; - cudaMalloc((void**)&d_I, n*sizeof(T)); + cudaMalloc((void**)&d_I, n*sizeof(int)); T* d_A = NULL; cudaMalloc((void**)&d_A, n*n*sizeof(T)); @@ -192,7 +192,7 @@ inline int lu_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[], T b int info = 0; int* d_I = NULL; - cudaMalloc((void**)&d_I, n*sizeof(T)); + cudaMalloc((void**)&d_I, n*sizeof(int)); T* d_A = NULL; cudaMalloc((void**)&d_A, n*n*sizeof(T)); @@ -306,7 +306,7 @@ inline int cholesky_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[ } T* d_B = NULL; - cudaMalloc((void**)d_B, n*nrhs*sizeof(T)); + cudaMalloc((void**)&d_B, n*nrhs*sizeof(T)); cublasSetMatrix(n, nrhs, sizeof(T), b, n, d_B, n); potrs(solverHandle, CUBLAS_FILL_MODE_LOWER, n, nrhs, d_A, n, d_B, n, d_info); @@ -331,7 +331,7 @@ inline int cholesky_solve_factored(cusolverDnHandle_t solverHandle, int n, int n cublasSetMatrix(n, n, sizeof(T), a, n, d_A, n); T* d_B = NULL; - cudaMalloc((void**)d_B, n*nrhs*sizeof(T)); + cudaMalloc((void**)&d_B, n*nrhs*sizeof(T)); cublasSetMatrix(n, nrhs, sizeof(T), b, n, d_B, n); int* d_info = NULL; @@ -461,7 +461,7 @@ inline int svd_factor(cusolverDnHandle_t solverHandle, bool compute_vectors, int cudaMalloc((void**)&d_U, m*m*sizeof(T)); T* d_V = NULL; - cudaMalloc((void**)&d_V, n*m*sizeof(T)); + cudaMalloc((void**)&d_V, n*n*sizeof(T)); T* work = NULL; int lWork = 0; @@ -474,7 +474,6 @@ inline int svd_factor(cusolverDnHandle_t solverHandle, bool compute_vectors, int int* d_info = NULL; cudaMalloc((void**)&d_info, sizeof(int)); - char job = compute_vectors ? 'A' : 'N'; gesvd(solverHandle, job, job, m, n, d_A, m, d_S, d_U, m, d_V, n, work, lWork, rwork, d_info); cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); @@ -529,7 +528,7 @@ inline int complex_svd_factor(cusolverDnHandle_t solverHandle, bool compute_vect gesvd(solverHandle, job, job, m, n, d_A, m, d_S, d_U, m, d_V, n, work, lWork, rwork, d_info); cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); - cublasGetVector(dim_s, sizeof(T), d_S, 1, s_local, 1); + cublasGetVector(dim_s, sizeof(R), d_S, 1, s_local, 1); cublasGetMatrix(m, m, sizeof(T), d_U, m, u, m); cublasGetMatrix(n, n, sizeof(T), d_V, n, v, n); diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex.cs index bad3b091..5936f3e0 100644 --- a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex.cs +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex.cs @@ -317,7 +317,7 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda if (work != null) { - throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + throw new ArgumentException(Resources.UserWorkBufferNotSupported); } Solver(SafeNativeMethods.z_lu_inverse(_solverHandle, _blasHandle, order, a)); @@ -356,7 +356,7 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda if (work != null) { - throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + throw new ArgumentException(Resources.UserWorkBufferNotSupported); } BLAS(SafeNativeMethods.z_lu_inverse_factored(_blasHandle, order, a, ipiv)); @@ -677,7 +677,7 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda if (work != null) { - throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + throw new ArgumentException(Resources.UserWorkBufferNotSupported); } if (u.Length != rowsA*rowsA) @@ -695,7 +695,9 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda throw new ArgumentException(Resources.ArgumentArraysSameLength, "s"); } - Solver(SafeNativeMethods.z_svd_factor(_solverHandle, computeVectors, rowsA, columnsA, a, s, u, vt)); + if (columnsA > rowsA || !computeVectors) // see remarks http://docs.nvidia.com/cuda/cusolver/index.html#cuds-lt-t-gt-gesvd + base.SingularValueDecomposition(computeVectors, a, rowsA, columnsA, s, u, vt, new Complex[rowsA]); + else Solver(SafeNativeMethods.z_svd_factor(_solverHandle, computeVectors, rowsA, columnsA, a, s, u, vt)); } } } diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex32.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex32.cs index 3e06022c..97a2307b 100644 --- a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex32.cs +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex32.cs @@ -317,7 +317,7 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda if (work != null) { - throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + throw new ArgumentException(Resources.UserWorkBufferNotSupported); } Solver(SafeNativeMethods.c_lu_inverse(_solverHandle, _blasHandle, order, a)); @@ -356,7 +356,7 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda if (work != null) { - throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + throw new ArgumentException(Resources.UserWorkBufferNotSupported); } BLAS(SafeNativeMethods.c_lu_inverse_factored(_blasHandle, order, a, ipiv)); @@ -589,7 +589,7 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda throw new ArgumentException(Resources.ArgumentArraysSameLength, "s"); } - SingularValueDecomposition(computeVectors, a, rowsA, columnsA, s, u, vt); + SingularValueDecomposition(computeVectors, a, rowsA, columnsA, s, u, vt, null); } /// @@ -677,7 +677,7 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda if (work != null) { - throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + throw new ArgumentException(Resources.UserWorkBufferNotSupported); } if (u.Length != rowsA*rowsA) @@ -695,7 +695,9 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda throw new ArgumentException(Resources.ArgumentArraysSameLength, "s"); } - Solver(SafeNativeMethods.c_svd_factor(_solverHandle, computeVectors, rowsA, columnsA, a, s, u, vt)); + if (columnsA > rowsA || !computeVectors) // see remarks http://docs.nvidia.com/cuda/cusolver/index.html#cuds-lt-t-gt-gesvd + base.SingularValueDecomposition(computeVectors, a, rowsA, columnsA, s, u, vt, new Complex32[rowsA]); + else Solver(SafeNativeMethods.c_svd_factor(_solverHandle, computeVectors, rowsA, columnsA, a, s, u, vt)); } } } diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Double.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Double.cs index 8213d8a5..6f8e3ca4 100644 --- a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Double.cs +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Double.cs @@ -317,7 +317,7 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda if (work != null) { - throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + throw new ArgumentException(Resources.UserWorkBufferNotSupported); } Solver(SafeNativeMethods.d_lu_inverse(_solverHandle, _blasHandle, order, a)); @@ -356,7 +356,7 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda if (work != null) { - throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + throw new ArgumentException(Resources.UserWorkBufferNotSupported); } BLAS(SafeNativeMethods.d_lu_inverse_factored(_blasHandle, order, a, ipiv)); @@ -589,7 +589,7 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda throw new ArgumentException(Resources.ArgumentArraysSameLength, "s"); } - SingularValueDecomposition(computeVectors, a, rowsA, columnsA, s, u, vt); + SingularValueDecomposition(computeVectors, a, rowsA, columnsA, s, u, vt, null); } /// @@ -677,7 +677,7 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda if (work != null) { - throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + throw new ArgumentException(Resources.UserWorkBufferNotSupported); } if (u.Length != rowsA*rowsA) @@ -695,8 +695,9 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda throw new ArgumentException(Resources.ArgumentArraysSameLength, "s"); } - - Solver (SafeNativeMethods.d_svd_factor(_solverHandle, computeVectors, rowsA, columnsA, a, s, u, vt)); + if (columnsA > rowsA || !computeVectors) // see remarks http://docs.nvidia.com/cuda/cusolver/index.html#cuds-lt-t-gt-gesvd + base.SingularValueDecomposition(computeVectors, a, rowsA, columnsA, s, u, vt, new double[rowsA]); + else Solver (SafeNativeMethods.d_svd_factor(_solverHandle, computeVectors, rowsA, columnsA, a, s, u, vt)); } } } diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Single.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Single.cs index f964e286..02aef2cf 100644 --- a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Single.cs +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Single.cs @@ -317,7 +317,7 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda if (work != null) { - throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + throw new ArgumentException(Resources.UserWorkBufferNotSupported); } Solver(SafeNativeMethods.s_lu_inverse(_solverHandle, _blasHandle, order, a)); @@ -356,7 +356,7 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda if (work != null) { - throw new NotSupportedException(Resources.UserWorkBufferNotSupported); + throw new ArgumentException(Resources.UserWorkBufferNotSupported); } BLAS(SafeNativeMethods.s_lu_inverse_factored(_blasHandle, order, a, ipiv)); @@ -695,7 +695,9 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda throw new ArgumentException(Resources.ArgumentArraysSameLength, "s"); } - Solver(SafeNativeMethods.s_svd_factor(_solverHandle, computeVectors, rowsA, columnsA, a, s, u, vt)); + if (columnsA > rowsA || !computeVectors) // see remarks http://docs.nvidia.com/cuda/cusolver/index.html#cuds-lt-t-gt-gesvd + base.SingularValueDecomposition(computeVectors, a, rowsA, columnsA, s, u, vt, new float[rowsA]); + else Solver(SafeNativeMethods.s_svd_factor(_solverHandle, computeVectors, rowsA, columnsA, a, s, u, vt)); } } } diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs index fe45efa8..543866e4 100644 --- a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs @@ -168,13 +168,13 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda throw new Exception("Mapping error"); case 6: // CUSOLVER_STATUS_EXECUTION_FAILED - throw new Exception("Execution failed"); + throw new NonConvergenceException("Execution failed"); case 7: //CUSOLVER_STATUS_INTERNAL_ERROR throw new Exception("Internal error"); case 8: // CUSOLVER_STATUS_MATRIX_TYPE_NOT_SUPPORTED - throw new NotSupportedException("Matrix type not supported"); + throw new ArgumentException("Matrix type not supported"); case 9: // CUSOLVER_STATUS_NOT_SUPPORTED throw new NotSupportedException(); diff --git a/src/UnitTests/LinearAlgebraProviderTests/Complex/LinearAlgebraProviderTests.cs b/src/UnitTests/LinearAlgebraProviderTests/Complex/LinearAlgebraProviderTests.cs index 7508a520..d17a75cd 100644 --- a/src/UnitTests/LinearAlgebraProviderTests/Complex/LinearAlgebraProviderTests.cs +++ b/src/UnitTests/LinearAlgebraProviderTests/Complex/LinearAlgebraProviderTests.cs @@ -444,7 +444,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Complex var a = new Complex[matrix.RowCount*matrix.RowCount]; Array.Copy(matrix.Values, a, a.Length); +#if CUDA + Complex[] work = null; +#else var work = new Complex[matrix.RowCount]; +#endif Control.LinearAlgebraProvider.LUInverse(a, matrix.RowCount, work); AssertHelpers.AlmostEqualRelative(a[0], -0.454545454545454, 13); @@ -473,7 +477,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Complex Control.LinearAlgebraProvider.LUFactor(a, matrix.RowCount, ipiv); +#if CUDA + Complex[] work = null; +#else var work = new Complex[matrix.RowCount]; +#endif Control.LinearAlgebraProvider.LUInverseFactored(a, matrix.RowCount, ipiv, work); AssertHelpers.AlmostEqualRelative(a[0], -0.454545454545454, 13); @@ -1447,7 +1455,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Complex var s = new Complex[matrix.RowCount]; var u = new Complex[matrix.RowCount*matrix.RowCount]; var vt = new Complex[matrix.ColumnCount*matrix.ColumnCount]; +#if CUDA + Complex[] work = null; +#else var work = new Complex[100]; +#endif Control.LinearAlgebraProvider.SingularValueDecomposition(true, a, matrix.RowCount, matrix.ColumnCount, s, u, vt, work); @@ -1486,7 +1498,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Complex var s = new Complex[matrix.ColumnCount]; var u = new Complex[matrix.RowCount*matrix.RowCount]; var vt = new Complex[matrix.ColumnCount*matrix.ColumnCount]; +#if CUDA + Complex[] work = null; +#else var work = new Complex[100]; +#endif Control.LinearAlgebraProvider.SingularValueDecomposition(true, a, matrix.RowCount, matrix.ColumnCount, s, u, vt, work); @@ -1522,7 +1538,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Complex var s = new Complex[matrix.RowCount]; var u = new Complex[matrix.RowCount*matrix.RowCount]; var vt = new Complex[matrix.ColumnCount*matrix.ColumnCount]; +#if CUDA + Complex[] work = null; +#else var work = new Complex[100]; +#endif Control.LinearAlgebraProvider.SingularValueDecomposition(true, a, matrix.RowCount, matrix.ColumnCount, s, u, vt, work); diff --git a/src/UnitTests/LinearAlgebraProviderTests/Complex32/LinearAlgebraProviderTests.cs b/src/UnitTests/LinearAlgebraProviderTests/Complex32/LinearAlgebraProviderTests.cs index d13ea874..d0b8a4e3 100644 --- a/src/UnitTests/LinearAlgebraProviderTests/Complex32/LinearAlgebraProviderTests.cs +++ b/src/UnitTests/LinearAlgebraProviderTests/Complex32/LinearAlgebraProviderTests.cs @@ -448,7 +448,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Complex32 var a = new Complex32[matrix.RowCount*matrix.RowCount]; Array.Copy(matrix.Values, a, a.Length); +#if CUDA + Complex32[] work = null; +#else var work = new Complex32[matrix.RowCount]; +#endif Control.LinearAlgebraProvider.LUInverse(a, matrix.RowCount, work); AssertHelpers.AlmostEqualRelative(a[0], -0.454545454545454f, 5); @@ -477,7 +481,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Complex32 Control.LinearAlgebraProvider.LUFactor(a, matrix.RowCount, ipiv); +#if CUDA + Complex32[] work = null; +#else var work = new Complex32[matrix.RowCount]; +#endif Control.LinearAlgebraProvider.LUInverseFactored(a, matrix.RowCount, ipiv, work); AssertHelpers.AlmostEqualRelative(a[0], -0.454545454545454f, 5); @@ -1452,7 +1460,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Complex32 var s = new Complex32[matrix.RowCount]; var u = new Complex32[matrix.RowCount*matrix.RowCount]; var vt = new Complex32[matrix.ColumnCount*matrix.ColumnCount]; +#if CUDA + Complex32[] work = null; +#else var work = new Complex32[100]; +#endif Control.LinearAlgebraProvider.SingularValueDecomposition(true, a, matrix.RowCount, matrix.ColumnCount, s, u, vt, work); @@ -1491,7 +1503,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Complex32 var s = new Complex32[matrix.ColumnCount]; var u = new Complex32[matrix.RowCount*matrix.RowCount]; var vt = new Complex32[matrix.ColumnCount*matrix.ColumnCount]; +#if CUDA + Complex32[] work = null; +#else var work = new Complex32[100]; +#endif Control.LinearAlgebraProvider.SingularValueDecomposition(true, a, matrix.RowCount, matrix.ColumnCount, s, u, vt, work); @@ -1527,7 +1543,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Complex32 var s = new Complex32[matrix.RowCount]; var u = new Complex32[matrix.RowCount*matrix.RowCount]; var vt = new Complex32[matrix.ColumnCount*matrix.ColumnCount]; +#if CUDA + Complex32[] work = null; +#else var work = new Complex32[100]; +#endif Control.LinearAlgebraProvider.SingularValueDecomposition(true, a, matrix.RowCount, matrix.ColumnCount, s, u, vt, work); diff --git a/src/UnitTests/LinearAlgebraProviderTests/Double/LinearAlgebraProviderTests.cs b/src/UnitTests/LinearAlgebraProviderTests/Double/LinearAlgebraProviderTests.cs index 2e50b594..6546bc55 100644 --- a/src/UnitTests/LinearAlgebraProviderTests/Double/LinearAlgebraProviderTests.cs +++ b/src/UnitTests/LinearAlgebraProviderTests/Double/LinearAlgebraProviderTests.cs @@ -438,7 +438,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Double var a = new double[matrix.RowCount*matrix.RowCount]; Array.Copy(matrix.Values, a, a.Length); +#if CUDA + double[] work = null; +#else var work = new double[matrix.RowCount]; +#endif Control.LinearAlgebraProvider.LUInverse(a, matrix.RowCount, work); AssertHelpers.AlmostEqualRelative(a[0], -0.454545454545454, 13); @@ -467,7 +471,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Double Control.LinearAlgebraProvider.LUFactor(a, matrix.RowCount, ipiv); +#if CUDA + double[] work = null; +#else var work = new double[matrix.RowCount]; +#endif Control.LinearAlgebraProvider.LUInverseFactored(a, matrix.RowCount, ipiv, work); AssertHelpers.AlmostEqualRelative(a[0], -0.454545454545454, 13); @@ -1442,7 +1450,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Double var s = new double[matrix.RowCount]; var u = new double[matrix.RowCount*matrix.RowCount]; var vt = new double[matrix.ColumnCount*matrix.ColumnCount]; +#if CUDA + double[] work = null; +#else var work = new double[100]; +#endif Control.LinearAlgebraProvider.SingularValueDecomposition(true, a, matrix.RowCount, matrix.ColumnCount, s, u, vt, work); @@ -1481,7 +1493,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Double var s = new double[matrix.ColumnCount]; var u = new double[matrix.RowCount*matrix.RowCount]; var vt = new double[matrix.ColumnCount*matrix.ColumnCount]; +#if CUDA + double[] work = null; +#else var work = new double[100]; +#endif Control.LinearAlgebraProvider.SingularValueDecomposition(true, a, matrix.RowCount, matrix.ColumnCount, s, u, vt, work); @@ -1517,7 +1533,11 @@ namespace MathNet.Numerics.UnitTests.LinearAlgebraProviderTests.Double var s = new double[matrix.RowCount]; var u = new double[matrix.RowCount*matrix.RowCount]; var vt = new double[matrix.ColumnCount*matrix.ColumnCount]; +#if CUDA + double[] work = null; +#else var work = new double[100]; +#endif Control.LinearAlgebraProvider.SingularValueDecomposition(true, a, matrix.RowCount, matrix.ColumnCount, s, u, vt, work); From fb30ee92f960f5f1ee6d1976bfe18d7264eb1e95 Mon Sep 17 00:00:00 2001 From: Matthew Johnson Date: Sun, 26 Apr 2015 19:52:42 +0100 Subject: [PATCH 09/14] Updating comments --- .../Cuda/CudaLinearAlgebraProvider.Complex.cs | 4 ++-- .../Cuda/CudaLinearAlgebraProvider.Complex32.cs | 4 ++-- .../Cuda/CudaLinearAlgebraProvider.Double.cs | 4 ++-- .../Cuda/CudaLinearAlgebraProvider.Single.cs | 4 ++-- .../LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs | 9 ++------- 5 files changed, 10 insertions(+), 15 deletions(-) diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex.cs index 5936f3e0..31abfbc5 100644 --- a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex.cs +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex.cs @@ -1,4 +1,4 @@ -// +// // Math.NET Numerics, part of the Math.NET Project // http://numerics.mathdotnet.com // http://github.com/mathnet/mathnet-numerics @@ -39,7 +39,7 @@ using MathNet.Numerics.Properties; namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda { /// - /// Intel's Math Kernel Library (MKL) linear algebra provider. + /// NVidia's CUDA Toolkit linear algebra provider. /// public partial class CudaLinearAlgebraProvider { diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex32.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex32.cs index 97a2307b..81679c15 100644 --- a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex32.cs +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Complex32.cs @@ -1,4 +1,4 @@ -// +// // Math.NET Numerics, part of the Math.NET Project // http://numerics.mathdotnet.com // http://github.com/mathnet/mathnet-numerics @@ -39,7 +39,7 @@ using MathNet.Numerics.Properties; namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda { /// - /// Intel's Math Kernel Library (MKL) linear algebra provider. + /// NVidia's CUDA Toolkit linear algebra provider. /// public partial class CudaLinearAlgebraProvider { diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Double.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Double.cs index 6f8e3ca4..6e73ed46 100644 --- a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Double.cs +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Double.cs @@ -1,4 +1,4 @@ -// +// // Math.NET Numerics, part of the Math.NET Project // http://numerics.mathdotnet.com // http://github.com/mathnet/mathnet-numerics @@ -39,7 +39,7 @@ using MathNet.Numerics.Properties; namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda { /// - /// Intel's Math Kernel Library (MKL) linear algebra provider. + /// NVidia's CUDA Toolkit linear algebra provider. /// public partial class CudaLinearAlgebraProvider { diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Single.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Single.cs index 02aef2cf..0ac4e34f 100644 --- a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Single.cs +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.Single.cs @@ -1,4 +1,4 @@ -// +// // Math.NET Numerics, part of the Math.NET Project // http://numerics.mathdotnet.com // http://github.com/mathnet/mathnet-numerics @@ -39,7 +39,7 @@ using MathNet.Numerics.Properties; namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda { /// - /// Intel's Math Kernel Library (MKL) linear algebra provider. + /// NVidia's CUDA Toolkit linear algebra provider. /// public partial class CudaLinearAlgebraProvider { diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs index 543866e4..afef619f 100644 --- a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs @@ -1,4 +1,4 @@ -// +// // Math.NET Numerics, part of the Math.NET Project // http://numerics.mathdotnet.com // http://github.com/mathnet/mathnet-numerics @@ -35,12 +35,7 @@ using System; namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda { /// - /// Consistency vs. performance trade-off between runs on different machines. - /// - - - /// - /// Intel's Math Kernel Library (MKL) linear algebra provider. + /// NVidia's CUDA Toolkit linear algebra provider. /// public partial class CudaLinearAlgebraProvider : ManagedLinearAlgebraProvider, IDisposable { From fd945979e44e691ae5404ac6c942f2f11c44a4b2 Mon Sep 17 00:00:00 2001 From: borfudin Date: Mon, 27 Apr 2015 11:58:55 +0100 Subject: [PATCH 10/14] Removing a temporary project and fixing a memory leak. --- MathNet.Numerics.NativeProviders.sln | 21 --------------------- src/NativeProviders/CUDA/lapack.cpp | 2 ++ 2 files changed, 2 insertions(+), 21 deletions(-) diff --git a/MathNet.Numerics.NativeProviders.sln b/MathNet.Numerics.NativeProviders.sln index 0297e6f8..0b0efacb 100644 --- a/MathNet.Numerics.NativeProviders.sln +++ b/MathNet.Numerics.NativeProviders.sln @@ -24,8 +24,6 @@ Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "CUDA", "src\NativeProviders EndProject Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "UnitTests-CUDA", "src\UnitTests\UnitTests-CUDA.csproj", "{E79C0395-01DC-4BC9-B86C-ED45790892C5}" EndProject -Project("{FAE04EC0-301F-11D3-BF4B-00C04F79EFBC}") = "Scratch", "Scratch\Scratch.csproj", "{2386FAD1-BB99-4597-885C-8EF81D0637BA}" -EndProject Global GlobalSection(SolutionConfigurationPlatforms) = preSolution Debug|Any CPU = Debug|Any CPU @@ -155,25 +153,6 @@ Global {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-Signed|Mixed Platforms.Build.0 = Release|Any CPU {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-Signed|Win32.ActiveCfg = Release|Any CPU {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-Signed|x64.ActiveCfg = Release|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Debug|Any CPU.ActiveCfg = Debug|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Debug|Any CPU.Build.0 = Debug|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Debug|Mixed Platforms.ActiveCfg = Debug|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Debug|Mixed Platforms.Build.0 = Debug|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Debug|Win32.ActiveCfg = Debug|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Debug|x64.ActiveCfg = Debug|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Debug|x64.Build.0 = Debug|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release|Any CPU.ActiveCfg = Release|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release|Any CPU.Build.0 = Release|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release|Mixed Platforms.ActiveCfg = Release|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release|Mixed Platforms.Build.0 = Release|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release|Win32.ActiveCfg = Release|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release|x64.ActiveCfg = Release|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release-Signed|Any CPU.ActiveCfg = Release|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release-Signed|Any CPU.Build.0 = Release|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release-Signed|Mixed Platforms.ActiveCfg = Release|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release-Signed|Mixed Platforms.Build.0 = Release|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release-Signed|Win32.ActiveCfg = Release|Any CPU - {2386FAD1-BB99-4597-885C-8EF81D0637BA}.Release-Signed|x64.ActiveCfg = Release|Any CPU EndGlobalSection GlobalSection(SolutionProperties) = preSolution HideSolutionNode = FALSE diff --git a/src/NativeProviders/CUDA/lapack.cpp b/src/NativeProviders/CUDA/lapack.cpp index b4cbf1e7..a7bbbfbd 100644 --- a/src/NativeProviders/CUDA/lapack.cpp +++ b/src/NativeProviders/CUDA/lapack.cpp @@ -209,6 +209,8 @@ inline int lu_solve(cusolverDnHandle_t solverHandle, int n, int nrhs, T a[], T b getrf(solverHandle, n, n, d_A, n, work, d_I, d_info); cudaMemcpy(&info, d_info, sizeof(int), cudaMemcpyDeviceToHost); + cudaFree(work); + if (info != 0) { cudaFree(d_I); From d29b1f0d427c57d18e591cc1902320c8cafe9fb8 Mon Sep 17 00:00:00 2001 From: borfudin Date: Mon, 27 Apr 2015 12:41:52 +0100 Subject: [PATCH 11/14] Fixing a comment warning. --- .../LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs | 10 ++++------ 1 file changed, 4 insertions(+), 6 deletions(-) diff --git a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs index afef619f..fa11f02e 100644 --- a/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs +++ b/src/Numerics/Providers/LinearAlgebra/Cuda/CudaLinearAlgebraProvider.cs @@ -46,12 +46,10 @@ namespace MathNet.Numerics.Providers.LinearAlgebra.Cuda private IntPtr _blasHandle; private IntPtr _solverHandle; - /// - /// Sets the desired bit consistency on repeated identical computations on varying CPU architectures, - /// as a trade-off with performance. - /// - /// VML optimal precision and rounding. - /// VML accuracy mode. + + /// + /// Constructor. + /// [CLSCompliant(false)] public CudaLinearAlgebraProvider() { From c0a94944d6d9c332138d0329f8e2109451f168b6 Mon Sep 17 00:00:00 2001 From: Christoph Ruegg Date: Thu, 30 Apr 2015 20:58:29 +0200 Subject: [PATCH 12/14] Native: uniform test compilation symbols --- src/UnitTests/UnitTests-CUDA.csproj | 4 ++-- src/UnitTests/UnitTests-MKL.csproj | 4 ++-- src/UnitTests/UseLinearAlgebraProvider.cs | 5 +++-- 3 files changed, 7 insertions(+), 6 deletions(-) diff --git a/src/UnitTests/UnitTests-CUDA.csproj b/src/UnitTests/UnitTests-CUDA.csproj index 598ce9a2..074df81c 100644 --- a/src/UnitTests/UnitTests-CUDA.csproj +++ b/src/UnitTests/UnitTests-CUDA.csproj @@ -16,7 +16,7 @@ ..\..\ - TRACE;CUDA + TRACE;NATIVE;CUDA ..\..\out\CUDA\Windows\ ..\..\obj\CUDA\Windows\x86\ ..\..\obj\CUDA\Windows\x86\ @@ -28,7 +28,7 @@ AnyCPU - TRACE;DEBUG;CUDA + TRACE;DEBUG;NATIVE;CUDA ..\..\out\CUDA\Windows\ ..\..\obj\CUDA\Windows\x86\ ..\..\obj\CUDA\Windows\x86\ diff --git a/src/UnitTests/UnitTests-MKL.csproj b/src/UnitTests/UnitTests-MKL.csproj index bbd73bbf..07bd43ec 100644 --- a/src/UnitTests/UnitTests-MKL.csproj +++ b/src/UnitTests/UnitTests-MKL.csproj @@ -16,7 +16,7 @@ ..\..\ - TRACE;NATIVE + TRACE;NATIVE;MKL ..\..\out\MKL\Windows\ ..\..\obj\MKL\Windows\x86\ ..\..\obj\MKL\Windows\x86\ @@ -28,7 +28,7 @@ AnyCPU - TRACE;DEBUG;NATIVE + TRACE;DEBUG;NATIVE;MKL ..\..\out\MKL\Windows\ ..\..\obj\MKL\Windows\x86\ ..\..\obj\MKL\Windows\x86\ diff --git a/src/UnitTests/UseLinearAlgebraProvider.cs b/src/UnitTests/UseLinearAlgebraProvider.cs index d45957e8..00266687 100644 --- a/src/UnitTests/UseLinearAlgebraProvider.cs +++ b/src/UnitTests/UseLinearAlgebraProvider.cs @@ -39,10 +39,11 @@ namespace MathNet.Numerics.UnitTests public void BeforeTest(TestDetails testDetails) { #if !NET35 && NATIVE +#if MKL Control.UseNativeMKL(); -#endif -#if CUDA +#elif CUDA Control.UseNativeCUDA(); +#endif #endif } From f58f6a3fff0b187a91246262cb02d7cf5d3359ef Mon Sep 17 00:00:00 2001 From: Christoph Ruegg Date: Thu, 30 Apr 2015 21:24:44 +0200 Subject: [PATCH 13/14] Native: split native provider resources to allow individual versioning --- src/NativeProviders/ATLAS/resource.h | 14 +++ src/NativeProviders/ATLAS/resource.rc | 101 ++++++++++++++++++ src/NativeProviders/CUDA/resource.h | 14 +++ src/NativeProviders/CUDA/resource.rc | 101 ++++++++++++++++++ src/NativeProviders/MKL/resource.h | 14 +++ src/NativeProviders/MKL/resource.rc | 101 ++++++++++++++++++ .../Windows/ATLAS/ATLASWrapper.vcxproj | 2 +- .../ATLAS/ATLASWrapper.vcxproj.filters | 2 +- .../Windows/ATLASEx/ATLASWrapper.vcproj | 2 +- .../Windows/ATLASEx/ATLASWrapper.vcxproj | 2 +- .../ATLASEx/ATLASWrapper.vcxproj.filters | 2 +- .../Windows/CUDA/CUDAWrapper.vcxproj | 2 +- .../Windows/CUDA/CUDAWrapper.vcxproj.filters | 2 +- .../Windows/MKL/MKLWrapper.vcproj | 2 +- .../Windows/MKL/MKLWrapper.vcxproj | 2 +- .../Windows/MKL/MKLWrapper.vcxproj.filters | 2 +- 16 files changed, 355 insertions(+), 10 deletions(-) create mode 100644 src/NativeProviders/ATLAS/resource.h create mode 100644 src/NativeProviders/ATLAS/resource.rc create mode 100644 src/NativeProviders/CUDA/resource.h create mode 100644 src/NativeProviders/CUDA/resource.rc create mode 100644 src/NativeProviders/MKL/resource.h create mode 100644 src/NativeProviders/MKL/resource.rc diff --git a/src/NativeProviders/ATLAS/resource.h b/src/NativeProviders/ATLAS/resource.h new file mode 100644 index 00000000..27e2900c --- /dev/null +++ b/src/NativeProviders/ATLAS/resource.h @@ -0,0 +1,14 @@ +//{{NO_DEPENDENCIES}} +// Microsoft Visual C++ generated include file. +// Used by resource.rc + +// Next default values for new objects +// +#ifdef APSTUDIO_INVOKED +#ifndef APSTUDIO_READONLY_SYMBOLS +#define _APS_NEXT_RESOURCE_VALUE 101 +#define _APS_NEXT_COMMAND_VALUE 40001 +#define _APS_NEXT_CONTROL_VALUE 1001 +#define _APS_NEXT_SYMED_VALUE 101 +#endif +#endif diff --git a/src/NativeProviders/ATLAS/resource.rc b/src/NativeProviders/ATLAS/resource.rc new file mode 100644 index 00000000..b49bd076 --- /dev/null +++ b/src/NativeProviders/ATLAS/resource.rc @@ -0,0 +1,101 @@ +// Microsoft Visual C++ generated resource script. +// +#include "resource.h" + +#define APSTUDIO_READONLY_SYMBOLS +///////////////////////////////////////////////////////////////////////////// +// +// Generated from the TEXTINCLUDE 2 resource. +// +#include "windows.h" + +///////////////////////////////////////////////////////////////////////////// +#undef APSTUDIO_READONLY_SYMBOLS + +///////////////////////////////////////////////////////////////////////////// +// English (United States) resources + +#if !defined(AFX_RESOURCE_DLL) || defined(AFX_TARG_ENU) +LANGUAGE LANG_ENGLISH, SUBLANG_ENGLISH_US +#pragma code_page(1252) + +#ifdef APSTUDIO_INVOKED +///////////////////////////////////////////////////////////////////////////// +// +// TEXTINCLUDE +// + +1 TEXTINCLUDE +BEGIN + "resource.h\0" +END + +2 TEXTINCLUDE +BEGIN + "#include ""windows.h""\r\n" + "\0" +END + +3 TEXTINCLUDE +BEGIN + "\r\n" + "\0" +END + +#endif // APSTUDIO_INVOKED + + +///////////////////////////////////////////////////////////////////////////// +// +// Version +// + +VS_VERSION_INFO VERSIONINFO + FILEVERSION 0,1,0,0 + PRODUCTVERSION 0,1,0,0 + FILEFLAGSMASK 0x17L +#ifdef _DEBUG + FILEFLAGS 0x1L +#else + FILEFLAGS 0x0L +#endif + FILEOS 0x4L + FILETYPE 0x2L + FILESUBTYPE 0x0L +BEGIN + BLOCK "StringFileInfo" + BEGIN + BLOCK "040904b0" + BEGIN + VALUE "Comments", "http://numerics.mathdotnet.com/" + VALUE "CompanyName", "Math.NET" + VALUE "FileDescription", "MathNET Numerics ATLAS Native Provider" + VALUE "FileVersion", "0.1.0.0" + VALUE "InternalName", "Math.NET" + VALUE "LegalCopyright", "Copyright (C) Math.NET 2009-2015" + VALUE "OriginalFilename", "MathNet.Numerics.ATLAS" + VALUE "ProductName", "Math.NET Numerics" + VALUE "ProductVersion", "0.1.0.0" + END + END + BLOCK "VarFileInfo" + BEGIN + VALUE "Translation", 0x409, 1200 + END +END + +#endif // English (United States) resources +///////////////////////////////////////////////////////////////////////////// + + + +#ifndef APSTUDIO_INVOKED +///////////////////////////////////////////////////////////////////////////// +// +// Generated from the TEXTINCLUDE 3 resource. +// + + +///////////////////////////////////////////////////////////////////////////// +#endif // not APSTUDIO_INVOKED + diff --git a/src/NativeProviders/CUDA/resource.h b/src/NativeProviders/CUDA/resource.h new file mode 100644 index 00000000..27e2900c --- /dev/null +++ b/src/NativeProviders/CUDA/resource.h @@ -0,0 +1,14 @@ +//{{NO_DEPENDENCIES}} +// Microsoft Visual C++ generated include file. +// Used by resource.rc + +// Next default values for new objects +// +#ifdef APSTUDIO_INVOKED +#ifndef APSTUDIO_READONLY_SYMBOLS +#define _APS_NEXT_RESOURCE_VALUE 101 +#define _APS_NEXT_COMMAND_VALUE 40001 +#define _APS_NEXT_CONTROL_VALUE 1001 +#define _APS_NEXT_SYMED_VALUE 101 +#endif +#endif diff --git a/src/NativeProviders/CUDA/resource.rc b/src/NativeProviders/CUDA/resource.rc new file mode 100644 index 00000000..30aa52c5 --- /dev/null +++ b/src/NativeProviders/CUDA/resource.rc @@ -0,0 +1,101 @@ +// Microsoft Visual C++ generated resource script. +// +#include "resource.h" + +#define APSTUDIO_READONLY_SYMBOLS +///////////////////////////////////////////////////////////////////////////// +// +// Generated from the TEXTINCLUDE 2 resource. +// +#include "windows.h" + +///////////////////////////////////////////////////////////////////////////// +#undef APSTUDIO_READONLY_SYMBOLS + +///////////////////////////////////////////////////////////////////////////// +// English (United States) resources + +#if !defined(AFX_RESOURCE_DLL) || defined(AFX_TARG_ENU) +LANGUAGE LANG_ENGLISH, SUBLANG_ENGLISH_US +#pragma code_page(1252) + +#ifdef APSTUDIO_INVOKED +///////////////////////////////////////////////////////////////////////////// +// +// TEXTINCLUDE +// + +1 TEXTINCLUDE +BEGIN + "resource.h\0" +END + +2 TEXTINCLUDE +BEGIN + "#include ""windows.h""\r\n" + "\0" +END + +3 TEXTINCLUDE +BEGIN + "\r\n" + "\0" +END + +#endif // APSTUDIO_INVOKED + + +///////////////////////////////////////////////////////////////////////////// +// +// Version +// + +VS_VERSION_INFO VERSIONINFO + FILEVERSION 0,1,0,0 + PRODUCTVERSION 0,1,0,0 + FILEFLAGSMASK 0x17L +#ifdef _DEBUG + FILEFLAGS 0x1L +#else + FILEFLAGS 0x0L +#endif + FILEOS 0x4L + FILETYPE 0x2L + FILESUBTYPE 0x0L +BEGIN + BLOCK "StringFileInfo" + BEGIN + BLOCK "040904b0" + BEGIN + VALUE "Comments", "http://numerics.mathdotnet.com/" + VALUE "CompanyName", "Math.NET" + VALUE "FileDescription", "MathNET Numerics CUDA Native Provider" + VALUE "FileVersion", "0.1.0.0" + VALUE "InternalName", "Math.NET" + VALUE "LegalCopyright", "Copyright (C) Math.NET 2009-2015" + VALUE "OriginalFilename", "MathNet.Numerics.CUDA" + VALUE "ProductName", "Math.NET Numerics" + VALUE "ProductVersion", "0.1.0.0" + END + END + BLOCK "VarFileInfo" + BEGIN + VALUE "Translation", 0x409, 1200 + END +END + +#endif // English (United States) resources +///////////////////////////////////////////////////////////////////////////// + + + +#ifndef APSTUDIO_INVOKED +///////////////////////////////////////////////////////////////////////////// +// +// Generated from the TEXTINCLUDE 3 resource. +// + + +///////////////////////////////////////////////////////////////////////////// +#endif // not APSTUDIO_INVOKED + diff --git a/src/NativeProviders/MKL/resource.h b/src/NativeProviders/MKL/resource.h new file mode 100644 index 00000000..27e2900c --- /dev/null +++ b/src/NativeProviders/MKL/resource.h @@ -0,0 +1,14 @@ +//{{NO_DEPENDENCIES}} +// Microsoft Visual C++ generated include file. +// Used by resource.rc + +// Next default values for new objects +// +#ifdef APSTUDIO_INVOKED +#ifndef APSTUDIO_READONLY_SYMBOLS +#define _APS_NEXT_RESOURCE_VALUE 101 +#define _APS_NEXT_COMMAND_VALUE 40001 +#define _APS_NEXT_CONTROL_VALUE 1001 +#define _APS_NEXT_SYMED_VALUE 101 +#endif +#endif diff --git a/src/NativeProviders/MKL/resource.rc b/src/NativeProviders/MKL/resource.rc new file mode 100644 index 00000000..6604b1a7 --- /dev/null +++ b/src/NativeProviders/MKL/resource.rc @@ -0,0 +1,101 @@ +// Microsoft Visual C++ generated resource script. +// +#include "resource.h" + +#define APSTUDIO_READONLY_SYMBOLS +///////////////////////////////////////////////////////////////////////////// +// +// Generated from the TEXTINCLUDE 2 resource. +// +#include "windows.h" + +///////////////////////////////////////////////////////////////////////////// +#undef APSTUDIO_READONLY_SYMBOLS + +///////////////////////////////////////////////////////////////////////////// +// English (United States) resources + +#if !defined(AFX_RESOURCE_DLL) || defined(AFX_TARG_ENU) +LANGUAGE LANG_ENGLISH, SUBLANG_ENGLISH_US +#pragma code_page(1252) + +#ifdef APSTUDIO_INVOKED +///////////////////////////////////////////////////////////////////////////// +// +// TEXTINCLUDE +// + +1 TEXTINCLUDE +BEGIN + "resource.h\0" +END + +2 TEXTINCLUDE +BEGIN + "#include ""windows.h""\r\n" + "\0" +END + +3 TEXTINCLUDE +BEGIN + "\r\n" + "\0" +END + +#endif // APSTUDIO_INVOKED + + +///////////////////////////////////////////////////////////////////////////// +// +// Version +// + +VS_VERSION_INFO VERSIONINFO + FILEVERSION 1,7,0,0 + PRODUCTVERSION 1,7,0,0 + FILEFLAGSMASK 0x17L +#ifdef _DEBUG + FILEFLAGS 0x1L +#else + FILEFLAGS 0x0L +#endif + FILEOS 0x4L + FILETYPE 0x2L + FILESUBTYPE 0x0L +BEGIN + BLOCK "StringFileInfo" + BEGIN + BLOCK "040904b0" + BEGIN + VALUE "Comments", "http://numerics.mathdotnet.com/" + VALUE "CompanyName", "Math.NET" + VALUE "FileDescription", "MathNET Numerics MKL Native Provider" + VALUE "FileVersion", "1.7.0.0" + VALUE "InternalName", "Math.NET" + VALUE "LegalCopyright", "Copyright (C) Math.NET 2009-2015" + VALUE "OriginalFilename", "MathNet.Numerics.MKL" + VALUE "ProductName", "Math.NET Numerics" + VALUE "ProductVersion", "1.7.0.0" + END + END + BLOCK "VarFileInfo" + BEGIN + VALUE "Translation", 0x409, 1200 + END +END + +#endif // English (United States) resources +///////////////////////////////////////////////////////////////////////////// + + + +#ifndef APSTUDIO_INVOKED +///////////////////////////////////////////////////////////////////////////// +// +// Generated from the TEXTINCLUDE 3 resource. +// + + +///////////////////////////////////////////////////////////////////////////// +#endif // not APSTUDIO_INVOKED + diff --git a/src/NativeProviders/Windows/ATLAS/ATLASWrapper.vcxproj b/src/NativeProviders/Windows/ATLAS/ATLASWrapper.vcxproj index 1b2951aa..f347fa23 100644 --- a/src/NativeProviders/Windows/ATLAS/ATLASWrapper.vcxproj +++ b/src/NativeProviders/Windows/ATLAS/ATLASWrapper.vcxproj @@ -148,7 +148,7 @@ - + diff --git a/src/NativeProviders/Windows/ATLAS/ATLASWrapper.vcxproj.filters b/src/NativeProviders/Windows/ATLAS/ATLASWrapper.vcxproj.filters index d6464ae4..518bf6df 100644 --- a/src/NativeProviders/Windows/ATLAS/ATLASWrapper.vcxproj.filters +++ b/src/NativeProviders/Windows/ATLAS/ATLASWrapper.vcxproj.filters @@ -15,7 +15,7 @@ - + Resource Files diff --git a/src/NativeProviders/Windows/ATLASEx/ATLASWrapper.vcproj b/src/NativeProviders/Windows/ATLASEx/ATLASWrapper.vcproj index d38c82fd..64823410 100644 --- a/src/NativeProviders/Windows/ATLASEx/ATLASWrapper.vcproj +++ b/src/NativeProviders/Windows/ATLASEx/ATLASWrapper.vcproj @@ -206,7 +206,7 @@ UniqueIdentifier="{67DA6AB6-F800-4c08-8B7A-83BB121AAD01}" > diff --git a/src/NativeProviders/Windows/ATLASEx/ATLASWrapper.vcxproj b/src/NativeProviders/Windows/ATLASEx/ATLASWrapper.vcxproj index 11109832..5c811855 100644 --- a/src/NativeProviders/Windows/ATLASEx/ATLASWrapper.vcxproj +++ b/src/NativeProviders/Windows/ATLASEx/ATLASWrapper.vcxproj @@ -15,7 +15,7 @@ - + diff --git a/src/NativeProviders/Windows/ATLASEx/ATLASWrapper.vcxproj.filters b/src/NativeProviders/Windows/ATLASEx/ATLASWrapper.vcxproj.filters index 58436d58..1cfd00b1 100644 --- a/src/NativeProviders/Windows/ATLASEx/ATLASWrapper.vcxproj.filters +++ b/src/NativeProviders/Windows/ATLASEx/ATLASWrapper.vcxproj.filters @@ -23,7 +23,7 @@ - + Resource Files diff --git a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj index 393b8c88..45f99146 100644 --- a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj +++ b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj @@ -19,7 +19,7 @@ - + diff --git a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters index 3c9be8c5..c67ec930 100644 --- a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters +++ b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj.filters @@ -15,7 +15,7 @@ - + Resource Files diff --git a/src/NativeProviders/Windows/MKL/MKLWrapper.vcproj b/src/NativeProviders/Windows/MKL/MKLWrapper.vcproj index 0ce61789..7b46f402 100644 --- a/src/NativeProviders/Windows/MKL/MKLWrapper.vcproj +++ b/src/NativeProviders/Windows/MKL/MKLWrapper.vcproj @@ -368,7 +368,7 @@ UniqueIdentifier="{67DA6AB6-F800-4c08-8B7A-83BB121AAD01}" > diff --git a/src/NativeProviders/Windows/MKL/MKLWrapper.vcxproj b/src/NativeProviders/Windows/MKL/MKLWrapper.vcxproj index 26d4874d..b2d46580 100644 --- a/src/NativeProviders/Windows/MKL/MKLWrapper.vcxproj +++ b/src/NativeProviders/Windows/MKL/MKLWrapper.vcxproj @@ -297,7 +297,7 @@ - + diff --git a/src/NativeProviders/Windows/MKL/MKLWrapper.vcxproj.filters b/src/NativeProviders/Windows/MKL/MKLWrapper.vcxproj.filters index 9ccf129e..2a00620e 100644 --- a/src/NativeProviders/Windows/MKL/MKLWrapper.vcxproj.filters +++ b/src/NativeProviders/Windows/MKL/MKLWrapper.vcxproj.filters @@ -35,7 +35,7 @@ - + Resource Files From d96e72df1a562410428c24676db39fc3cf9ac8b6 Mon Sep 17 00:00:00 2001 From: Christoph Ruegg Date: Sun, 3 May 2015 22:37:21 +0200 Subject: [PATCH 14/14] Native: prepare build integration for CUDA native provider --- MathNet.Numerics.NativeProviders.sln | 84 +++++++++++++++++ MathNet.Numerics.sln | 3 + RELEASENOTES-CUDA.md | 3 + RELEASENOTES-OpenBLAS.md | 2 + build.fsx | 94 ++++++++++++++++--- build/MathNet.Numerics.CUDA.Win.targets | 74 +++++++++++++++ build/MathNet.Numerics.MKL.Win.targets | 74 +++++++++++++++ .../Windows/CUDA/CUDAWrapper.vcxproj | 4 +- 8 files changed, 321 insertions(+), 17 deletions(-) create mode 100644 RELEASENOTES-CUDA.md create mode 100644 RELEASENOTES-OpenBLAS.md create mode 100644 build/MathNet.Numerics.CUDA.Win.targets create mode 100644 build/MathNet.Numerics.MKL.Win.targets diff --git a/MathNet.Numerics.NativeProviders.sln b/MathNet.Numerics.NativeProviders.sln index 0b0efacb..a83444e8 100644 --- a/MathNet.Numerics.NativeProviders.sln +++ b/MathNet.Numerics.NativeProviders.sln @@ -34,6 +34,14 @@ Global Release|Mixed Platforms = Release|Mixed Platforms Release|Win32 = Release|Win32 Release|x64 = Release|x64 + Release-CUDA|Any CPU = Release-CUDA|Any CPU + Release-CUDA|Mixed Platforms = Release-CUDA|Mixed Platforms + Release-CUDA|Win32 = Release-CUDA|Win32 + Release-CUDA|x64 = Release-CUDA|x64 + Release-MKL|Any CPU = Release-MKL|Any CPU + Release-MKL|Mixed Platforms = Release-MKL|Mixed Platforms + Release-MKL|Win32 = Release-MKL|Win32 + Release-MKL|x64 = Release-MKL|x64 Release-Signed|Any CPU = Release-Signed|Any CPU Release-Signed|Mixed Platforms = Release-Signed|Mixed Platforms Release-Signed|Win32 = Release-Signed|Win32 @@ -54,6 +62,18 @@ Global {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release|Win32.Build.0 = Release|Win32 {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release|x64.ActiveCfg = Release|x64 {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release|x64.Build.0 = Release|x64 + {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-CUDA|Any CPU.ActiveCfg = Release|Win32 + {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-CUDA|Mixed Platforms.ActiveCfg = Release|Win32 + {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-CUDA|Mixed Platforms.Build.0 = Release|Win32 + {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-CUDA|Win32.ActiveCfg = Release|Win32 + {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-CUDA|x64.ActiveCfg = Release|x64 + {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-MKL|Any CPU.ActiveCfg = Release|Win32 + {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-MKL|Mixed Platforms.ActiveCfg = Release|Win32 + {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-MKL|Mixed Platforms.Build.0 = Release|Win32 + {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-MKL|Win32.ActiveCfg = Release|Win32 + {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-MKL|Win32.Build.0 = Release|Win32 + {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-MKL|x64.ActiveCfg = Release|x64 + {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-MKL|x64.Build.0 = Release|x64 {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-Signed|Any CPU.ActiveCfg = Release|Win32 {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-Signed|Mixed Platforms.ActiveCfg = Release|Win32 {C0B0DBA9-7FB0-4C87-BDB1-3EED19DC2B8F}.Release-Signed|Mixed Platforms.Build.0 = Release|Win32 @@ -69,6 +89,14 @@ Global {2362B8AC-C52B-45E4-A1BF-C682A4DB4220}.Release|Mixed Platforms.ActiveCfg = Release|Win32 {2362B8AC-C52B-45E4-A1BF-C682A4DB4220}.Release|Win32.ActiveCfg = Release|Win32 {2362B8AC-C52B-45E4-A1BF-C682A4DB4220}.Release|x64.ActiveCfg = Release|x64 + {2362B8AC-C52B-45E4-A1BF-C682A4DB4220}.Release-CUDA|Any CPU.ActiveCfg = Release|Win32 + {2362B8AC-C52B-45E4-A1BF-C682A4DB4220}.Release-CUDA|Mixed Platforms.ActiveCfg = Release|Win32 + {2362B8AC-C52B-45E4-A1BF-C682A4DB4220}.Release-CUDA|Win32.ActiveCfg = Release|Win32 + {2362B8AC-C52B-45E4-A1BF-C682A4DB4220}.Release-CUDA|x64.ActiveCfg = Release|x64 + {2362B8AC-C52B-45E4-A1BF-C682A4DB4220}.Release-MKL|Any CPU.ActiveCfg = Release|Win32 + {2362B8AC-C52B-45E4-A1BF-C682A4DB4220}.Release-MKL|Mixed Platforms.ActiveCfg = Release|Win32 + {2362B8AC-C52B-45E4-A1BF-C682A4DB4220}.Release-MKL|Win32.ActiveCfg = Release|Win32 + {2362B8AC-C52B-45E4-A1BF-C682A4DB4220}.Release-MKL|x64.ActiveCfg = Release|x64 {2362B8AC-C52B-45E4-A1BF-C682A4DB4220}.Release-Signed|Any CPU.ActiveCfg = Release|Win32 {2362B8AC-C52B-45E4-A1BF-C682A4DB4220}.Release-Signed|Mixed Platforms.ActiveCfg = Release|Win32 {2362B8AC-C52B-45E4-A1BF-C682A4DB4220}.Release-Signed|Win32.ActiveCfg = Release|Win32 @@ -91,6 +119,22 @@ Global {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release|Win32.Build.0 = Release|Any CPU {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release|x64.ActiveCfg = Release|Any CPU {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release|x64.Build.0 = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-CUDA|Any CPU.ActiveCfg = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-CUDA|Any CPU.Build.0 = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-CUDA|Mixed Platforms.ActiveCfg = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-CUDA|Mixed Platforms.Build.0 = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-CUDA|Win32.ActiveCfg = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-CUDA|Win32.Build.0 = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-CUDA|x64.ActiveCfg = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-CUDA|x64.Build.0 = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-MKL|Any CPU.ActiveCfg = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-MKL|Any CPU.Build.0 = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-MKL|Mixed Platforms.ActiveCfg = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-MKL|Mixed Platforms.Build.0 = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-MKL|Win32.ActiveCfg = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-MKL|Win32.Build.0 = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-MKL|x64.ActiveCfg = Release|Any CPU + {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-MKL|x64.Build.0 = Release|Any CPU {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-Signed|Any CPU.ActiveCfg = Release-Signed|Any CPU {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-Signed|Any CPU.Build.0 = Release-Signed|Any CPU {B7CAE5F4-A23F-4438-B5BE-41226618B695}.Release-Signed|Mixed Platforms.ActiveCfg = Release-Signed|Any CPU @@ -109,6 +153,20 @@ Global {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release|Mixed Platforms.Build.0 = Release|Any CPU {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release|Win32.ActiveCfg = Release|Any CPU {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release|x64.ActiveCfg = Release|Any CPU + {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-CUDA|Any CPU.ActiveCfg = Release|Any CPU + {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-CUDA|Any CPU.Build.0 = Release|Any CPU + {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-CUDA|Mixed Platforms.ActiveCfg = Release|Any CPU + {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-CUDA|Mixed Platforms.Build.0 = Release|Any CPU + {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-CUDA|Win32.ActiveCfg = Release|Any CPU + {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-CUDA|x64.ActiveCfg = Release|Any CPU + {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-MKL|Any CPU.ActiveCfg = Release|Any CPU + {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-MKL|Any CPU.Build.0 = Release|Any CPU + {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-MKL|Mixed Platforms.ActiveCfg = Release|Any CPU + {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-MKL|Mixed Platforms.Build.0 = Release|Any CPU + {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-MKL|Win32.ActiveCfg = Release|Any CPU + {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-MKL|Win32.Build.0 = Release|Any CPU + {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-MKL|x64.ActiveCfg = Release|Any CPU + {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-MKL|x64.Build.0 = Release|Any CPU {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-Signed|Any CPU.ActiveCfg = Release|Any CPU {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-Signed|Any CPU.Build.0 = Release|Any CPU {3515A344-AB5F-41C7-A14C-04A79B3FFAB1}.Release-Signed|Mixed Platforms.ActiveCfg = Release|Any CPU @@ -129,6 +187,18 @@ Global {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|Win32.Build.0 = Release|Win32 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|x64.ActiveCfg = Release|x64 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release|x64.Build.0 = Release|x64 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-CUDA|Any CPU.ActiveCfg = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-CUDA|Mixed Platforms.ActiveCfg = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-CUDA|Mixed Platforms.Build.0 = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-CUDA|Win32.ActiveCfg = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-CUDA|Win32.Build.0 = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-CUDA|x64.ActiveCfg = Release|x64 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-CUDA|x64.Build.0 = Release|x64 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-MKL|Any CPU.ActiveCfg = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-MKL|Mixed Platforms.ActiveCfg = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-MKL|Mixed Platforms.Build.0 = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-MKL|Win32.ActiveCfg = Release|Win32 + {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-MKL|x64.ActiveCfg = Release|x64 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|Any CPU.ActiveCfg = Release|Win32 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|Mixed Platforms.ActiveCfg = Release|Win32 {5A52B796-7F41-4C90-8DE2-F3F391C4482C}.Release-Signed|Mixed Platforms.Build.0 = Release|Win32 @@ -147,6 +217,20 @@ Global {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release|Mixed Platforms.Build.0 = Release|Any CPU {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release|Win32.ActiveCfg = Release|Any CPU {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release|x64.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-CUDA|Any CPU.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-CUDA|Any CPU.Build.0 = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-CUDA|Mixed Platforms.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-CUDA|Mixed Platforms.Build.0 = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-CUDA|Win32.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-CUDA|Win32.Build.0 = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-CUDA|x64.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-CUDA|x64.Build.0 = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-MKL|Any CPU.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-MKL|Any CPU.Build.0 = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-MKL|Mixed Platforms.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-MKL|Mixed Platforms.Build.0 = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-MKL|Win32.ActiveCfg = Release|Any CPU + {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-MKL|x64.ActiveCfg = Release|Any CPU {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-Signed|Any CPU.ActiveCfg = Release|Any CPU {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-Signed|Any CPU.Build.0 = Release|Any CPU {E79C0395-01DC-4BC9-B86C-ED45790892C5}.Release-Signed|Mixed Platforms.ActiveCfg = Release|Any CPU diff --git a/MathNet.Numerics.sln b/MathNet.Numerics.sln index 58787f08..4eb3bebe 100644 --- a/MathNet.Numerics.sln +++ b/MathNet.Numerics.sln @@ -9,8 +9,10 @@ Project("{2150E333-8FDC-42A3-9474-1A3956D46DE8}") = "Readme", "Readme", "{C2F374 CONTRIBUTORS.md = CONTRIBUTORS.md LICENSE.md = LICENSE.md README.md = README.md + RELEASENOTES-CUDA.md = RELEASENOTES-CUDA.md RELEASENOTES-Data.md = RELEASENOTES-Data.md RELEASENOTES-MKL.md = RELEASENOTES-MKL.md + RELEASENOTES-OpenBLAS.md = RELEASENOTES-OpenBLAS.md RELEASENOTES.md = RELEASENOTES.md EndProjectSection EndProject @@ -27,6 +29,7 @@ Project("{2150E333-8FDC-42A3-9474-1A3956D46DE8}") = "Build", "Build", "{A4A66FA9 docs\tools\build-docs.fsx = docs\tools\build-docs.fsx build.fsx = build.fsx build\MathNet.Numerics.Extension.nuspec = build\MathNet.Numerics.Extension.nuspec + build\MathNet.Numerics.MKL.Win.targets = build\MathNet.Numerics.MKL.Win.targets build\MathNet.Numerics.nuspec = build\MathNet.Numerics.nuspec paket.dependencies = paket.dependencies paket.lock = paket.lock diff --git a/RELEASENOTES-CUDA.md b/RELEASENOTES-CUDA.md new file mode 100644 index 00000000..0ab166d4 --- /dev/null +++ b/RELEASENOTES-CUDA.md @@ -0,0 +1,3 @@ +### 0.1.0-alpha - TBA +* With Nvidia CUDA 7.0.28 +* Initial version diff --git a/RELEASENOTES-OpenBLAS.md b/RELEASENOTES-OpenBLAS.md new file mode 100644 index 00000000..2951cc2c --- /dev/null +++ b/RELEASENOTES-OpenBLAS.md @@ -0,0 +1,2 @@ +### 0.1.0-alpha - TBA +* Initial version diff --git a/build.fsx b/build.fsx index 505d73fd..f0a7eade 100644 --- a/build.fsx +++ b/build.fsx @@ -51,6 +51,13 @@ let mklPackageVersion = mklRelease.NugetVersion let mklReleaseNotes = mklRelease.Notes |> List.map (fun l -> l.Replace("*","").Replace("`","")) |> toLines trace (sprintf " Math.NET Numerics MKL Provider v%s" mklPackageVersion) +let cudaRelease = LoadReleaseNotes "RELEASENOTES-CUDA.md" +let cudaBuildPart = "0" +let cudaAssemblyVersion = cudaRelease.AssemblyVersion + "." + cudaBuildPart +let cudaPackageVersion = cudaRelease.NugetVersion +let cudaReleaseNotes = cudaRelease.Notes |> List.map (fun l -> l.Replace("*","").Replace("`","")) |> toLines +trace (sprintf " Math.NET Numerics CUDA Provider v%s" cudaPackageVersion) + let dataRelease = LoadReleaseNotes "RELEASENOTES-Data.md" let dataBuildPart = "0" let dataAssemblyVersion = dataRelease.AssemblyVersion + "." + dataBuildPart @@ -187,10 +194,10 @@ let coreSignedBundle = // NATIVE PROVIDER PACKAGES -let mklWin32Pack = - { Id = "MathNet.Numerics.MKL.Win-x86" +let mklWinPack = + { Id = "MathNet.Numerics.MKL.Win" Version = mklPackageVersion - Title = "Math.NET Numerics - MKL Native Provider for Windows (x86)" + Title = "Math.NET Numerics - MKL Native Provider for Windows (x64 and x86)" Summary = "" Description = "Intel MKL native libraries for Math.NET Numerics. Requires an Intel MKL license if redistributed." ReleaseNotes = mklReleaseNotes @@ -198,15 +205,32 @@ let mklWin32Pack = Authors = [ "Christoph Ruegg"; "Marcus Cuda"; "Jurgen Van Gael" ] Dependencies = [ { FrameworkVersion="" - Dependencies=[ "MathNet.Numerics", "2.4.0" ] } ] + Dependencies=[ "MathNet.Numerics", "3.6.0" ] } ] Files = - [ @"..\..\out\MKL\Windows\x86\libiomp5md.dll", Some "content", None; - @"..\..\out\MKL\Windows\x86\MathNet.Numerics.MKL.dll", Some "content", None ] } + [ @"MathNet.Numerics.MKL.Win.targets", Some "build", None; + @"..\..\out\MKL\Windows\x64\libiomp5md.dll", Some "build\x64", None; + @"..\..\out\MKL\Windows\x64\MathNet.Numerics.MKL.dll", Some "build\x64", None; + @"..\..\out\MKL\Windows\x86\libiomp5md.dll", Some "build\x86", None; + @"..\..\out\MKL\Windows\x86\MathNet.Numerics.MKL.dll", Some "build\x86", None ] } + +let mklWin32Pack = + { mklWinPack with + Id = "MathNet.Numerics.MKL.Win-x86" + Title = "Math.NET Numerics - MKL Native Provider for Windows (x86)" + Dependencies = + [ { FrameworkVersion="" + Dependencies=[ "MathNet.Numerics", "2.4.0" ] } ] + Files = + [ @"..\..\out\MKL\Windows\x86\libiomp5md.dll", Some "content", None; + @"..\..\out\MKL\Windows\x86\MathNet.Numerics.MKL.dll", Some "content", None ] } let mklWin64Pack = - { mklWin32Pack with + { mklWinPack with Id = "MathNet.Numerics.MKL.Win-x64" Title = "Math.NET Numerics - MKL Native Provider for Windows (x64)" + Dependencies = + [ { FrameworkVersion="" + Dependencies=[ "MathNet.Numerics", "2.4.0" ] } ] Files = [ @"..\..\out\MKL\Windows\x64\libiomp5md.dll", Some "content", None; @"..\..\out\MKL\Windows\x64\MathNet.Numerics.MKL.dll", Some "content", None ] } @@ -241,7 +265,7 @@ let mklWinBundle = Title = "Math.NET Numerics MKL Native Provider for Windows" ReleaseNotesFile = "RELEASENOTES-MKL.md" FsLoader = false - Packages = [ mklWin32Pack; mklWin64Pack ] } + Packages = [ mklWinPack; mklWin32Pack; mklWin64Pack ] } let mklLinuxBundle = { Id = "MathNet.Numerics.MKL.Linux" @@ -251,6 +275,37 @@ let mklLinuxBundle = FsLoader = false Packages = [ mklLinux32Pack; mklLinux64Pack ] } +let cudaWinPack = + { Id = "MathNet.Numerics.CUDA.Win" + Version = cudaPackageVersion + Title = "Math.NET Numerics - CUDA Native Provider for Windows (x64 and x86)" + Summary = "" + Description = "Nvidia CUDA native libraries for Math.NET Numerics." + ReleaseNotes = cudaReleaseNotes + Tags = "math numeric statistics probability integration interpolation linear algebra matrix fft native cuda gpu" + Authors = [ "Matthew A Johnson"; "Christoph Ruegg" ] + Dependencies = + [ { FrameworkVersion="" + Dependencies=[ "MathNet.Numerics", "3.7.0" ] } ] + Files = + [ @"MathNet.Numerics.CUDA.Win.targets", Some "build", None; + @"..\..\out\CUDA\Windows\x64\cublas64_70.dll", Some "content", None; + @"..\..\out\CUDA\Windows\x64\cudart64_70.dll", Some "content", None; + @"..\..\out\CUDA\Windows\x64\cusolver64_70.dll", Some "content", None; + @"..\..\out\CUDA\Windows\x64\MathNet.Numerics.CUDA.dll", Some "content", None + @"..\..\out\CUDA\Windows\x86\cublas32_70.dll", Some "content", None; + @"..\..\out\CUDA\Windows\x86\cudart32_70.dll", Some "content", None; + @"..\..\out\CUDA\Windows\x86\cusolver32_70.dll", Some "content", None; + @"..\..\out\CUDA\Windows\x86\MathNet.Numerics.CUDA.dll", Some "content", None ] } + +let cudaWinBundle = + { Id = "MathNet.Numerics.CUDA.Win" + Version = mklPackageVersion + Title = "Math.NET Numerics CUDA Native Provider for Windows" + ReleaseNotesFile = "RELEASENOTES-CUDA.md" + FsLoader = false + Packages = [ cudaWinPack ] } + // DATA EXTENSION PACKAGES @@ -309,7 +364,7 @@ Target "Clean" (fun _ -> CleanDirs [ "out/lib-debug/Net35"; "out/lib-debug/Net40"; "out/lib-debug/Profile7"; "out/lib-debug/Profile47"; "out/lib-debug/Profile78"; "out/lib-debug/Profile259"; "out/lib-debug/Profile328" ] CleanDirs [ "out/test-debug/Net35"; "out/test-debug/Net40"; "out/test-debug/Profile7"; "out/test-debug/Profile47"; "out/test-debug/Profile78"; "out/test-debug/Profile259"; "out/test-debug/Profile328" ] CleanDirs [ "out/lib-signed/Net40"; "out/test-signed/Net40" ] // Signed Build - CleanDirs [ "out/MKL"; "out/ATLAS" ] // Native Providers + CleanDirs [ "out/MKL"; "out/ATLAS"; "out/CUDA"; "out/OpenBLAS" ] // Native Providers CleanDirs [ "out/Data" ]) // Data Extensions Target "ApplyVersion" (fun _ -> @@ -328,7 +383,11 @@ Target "ApplyVersion" (fun _ -> ReplaceInFile (regex_replace @"\d+\.\d+\.\d+\.\d+" mklAssemblyVersion >> regex_replace @"\d+,\d+,\d+,\d+" (replace "." "," mklAssemblyVersion)) - "src/NativeProviders/Common/resource.rc") + "src/NativeProviders/MKL/resource.rc" + ReplaceInFile + (regex_replace @"\d+\.\d+\.\d+\.\d+" cudaAssemblyVersion + >> regex_replace @"\d+,\d+,\d+,\d+" (replace "." "," cudaAssemblyVersion)) + "src/NativeProviders/CUDA/resource.rc") Target "Prepare" DoNothing "Start" @@ -344,8 +403,8 @@ Target "Prepare" DoNothing let buildConfig config subject = MSBuild "" (if hasBuildParam "incremental" then "Build" else "Rebuild") [ "Configuration", config ] subject |> ignore let build subject = buildConfig "Release" subject let buildSigned subject = buildConfig "Release-Signed" subject -let nativeWin32Build subject = MSBuild "" (if hasBuildParam "incremental" then "Build" else "Rebuild") [("Configuration","Release"); ("Platform","Win32")] subject |> ignore -let nativeWin64Build subject = MSBuild "" (if hasBuildParam "incremental" then "Build" else "Rebuild") [("Configuration","Release"); ("Platform","x64")] subject |> ignore +let buildConfig32 config subject = MSBuild "" (if hasBuildParam "incremental" then "Build" else "Rebuild") [("Configuration", config); ("Platform","Win32")] subject |> ignore +let buildConfig64 config subject = MSBuild "" (if hasBuildParam "incremental" then "Build" else "Rebuild") [("Configuration", config); ("Platform","x64")] subject |> ignore Target "BuildMain" (fun _ -> build !! "MathNet.Numerics.sln") Target "BuildNet35" (fun _ -> build !! "MathNet.Numerics.Net35Only.sln") @@ -360,13 +419,18 @@ Target "Build" DoNothing =?> ("BuildMain", not (hasBuildParam "all" || hasBuildParam "release" || hasBuildParam "net35" || hasBuildParam "signed")) ==> "Build" -Target "MklWin32Build" (fun _ -> nativeWin32Build !! "MathNet.Numerics.NativeProviders.sln") -Target "MklWin64Build" (fun _ -> nativeWin64Build !! "MathNet.Numerics.NativeProviders.sln") - +Target "MklWin32Build" (fun _ -> buildConfig32 "Release-MKL" !! "MathNet.Numerics.NativeProviders.sln") +Target "MklWin64Build" (fun _ -> buildConfig64 "Release-MKL" !! "MathNet.Numerics.NativeProviders.sln") Target "MklWinBuild" DoNothing "Prepare" ==> "MklWin32Build" ==> "MklWinBuild" "Prepare" ==> "MklWin64Build" ==> "MklWinBuild" +Target "CudaWin32Build" (fun _ -> buildConfig32 "Release-CUDA" !! "MathNet.Numerics.NativeProviders.sln") +Target "CudaWin64Build" (fun _ -> buildConfig64 "Release-CUDA" !! "MathNet.Numerics.NativeProviders.sln") +Target "CudaWinBuild" DoNothing +"Prepare" ==> "CudaWin32Build" ==> "CudaWinBuild" +"Prepare" ==> "CudaWin64Build" ==> "CudaWinBuild" + Target "DataBuild" (fun _ -> build !! "MathNet.Numerics.Data.sln") "Prepare" ==> "DataBuild" diff --git a/build/MathNet.Numerics.CUDA.Win.targets b/build/MathNet.Numerics.CUDA.Win.targets new file mode 100644 index 00000000..74be39a2 --- /dev/null +++ b/build/MathNet.Numerics.CUDA.Win.targets @@ -0,0 +1,74 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + $(BuildDependsOn); + CopyMathNetInteropFiles; + + + $(CleanDependsOn); + CleanMathNetInteropFiles; + + + diff --git a/build/MathNet.Numerics.MKL.Win.targets b/build/MathNet.Numerics.MKL.Win.targets new file mode 100644 index 00000000..870de2b4 --- /dev/null +++ b/build/MathNet.Numerics.MKL.Win.targets @@ -0,0 +1,74 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + $(BuildDependsOn); + CopyMathNetInteropFiles; + + + $(CleanDependsOn); + CleanMathNetInteropFiles; + + + diff --git a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj index 45f99146..476349ff 100644 --- a/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj +++ b/src/NativeProviders/Windows/CUDA/CUDAWrapper.vcxproj @@ -105,7 +105,7 @@ true cublas.lib;cublas_device.lib;%(AdditionalDependencies) - $(CUDA_PATH)\lib\x64;%(AdditionalLibraryDirectories) + $(CUDA_PATH)\lib\Win32;%(AdditionalLibraryDirectories) @@ -141,7 +141,7 @@ copy "$(CUDA_PATH)\bin\cudart64_70.dll" $(OutputPath) true true cublas.lib;%(AdditionalDependencies) - $(CUDA_PATH)\lib\x64;%(AdditionalLibraryDirectories) + $(CUDA_PATH)\lib\Win32;%(AdditionalLibraryDirectories)