Merge pull request #2 from CNugteren/xgemv

Added GEMV routine
author: Cedric Nugteren <web@cedricnugteren.nl> 2015-06-15 18:35:27 +0200
committer: Cedric Nugteren <web@cedricnugteren.nl> 2015-06-15 18:35:27 +0200
commit: 9e2fba9ab9cab1f94dfe143fc6e163f47b6d6f39 (patch)
tree: f4e556ee5bda4e634c7af1db094e11f6def4e07e /include
parent: 3c17c1c13313022879c8caf289d0f47ea5d7d22d (diff)
parent: f925d47dadb5f7dd161e44e5da2142734faa3176 (diff)
5 files changed, 200 insertions, 3 deletions
diff --git a/include/clblast.h b/include/clblast.h
index 4c3c5201..231348b8 100644
--- a/include/clblast.h
+++ b/include/clblast.h
@@ -85,7 +85,7 @@ enum class Precision { kHalf = 16, kSingle = 32, kDouble = 64,
 
 // Templated-precision vector-times-constant plus vector: SAXPY/DAXPY/CAXPY/ZAXPY
 template <typename T>
-StatusCode Axpy(const size_t m, const T alpha,
+StatusCode Axpy(const size_t n, const T alpha,
                 const cl_mem x_buffer, const size_t x_offset, const size_t x_inc,
                 cl_mem y_buffer, const size_t y_offset, const size_t y_inc,
                 cl_command_queue* queue, cl_event* event);
@@ -93,10 +93,21 @@ StatusCode Axpy(const size_t m, const T alpha,
 // =================================================================================================
 // BLAS level-2 (matrix-vector) routines
 
+// Templated-precision generalized matrix-vector multiplication: SGEMV/DGEMV/CGEMV/ZGEMV
+template <typename T>
+StatusCode Gemv(const Layout layout, const Transpose transpose_a,
+                const size_t m, const size_t n,
+                const T alpha,
+                const cl_mem a_buffer, const size_t a_offset, const size_t a_ld,
+                const cl_mem x_buffer, const size_t x_offset, const size_t x_inc,
+                const T beta,
+                cl_mem y_buffer, const size_t y_offset, const size_t y_inc,
+                cl_command_queue* queue, cl_event* event);
+
 // =================================================================================================
 // BLAS level-3 (matrix-matrix) routines
 
-// Templated-precision generalized matrix multiplication: SGEMM/DGEMM
+// Templated-precision generalized matrix-matrix multiplication: SGEMM/DGEMM
 template <typename T>
 StatusCode Gemm(const Layout layout, const Transpose transpose_a, const Transpose transpose_b,
                 const size_t m, const size_t n, const size_t k,
@@ -107,7 +118,7 @@ StatusCode Gemm(const Layout layout, const Transpose transpose_a, const Transpos
                 cl_mem c_buffer, const size_t c_offset, const size_t c_ld,
                 cl_command_queue* queue, cl_event* event);
 
-// Templated-precision symmetric matrix multiplication: SSYMM/DSYMM
+// Templated-precision symmetric matrix-matrix multiplication: SSYMM/DSYMM
 template <typename T>
 StatusCode Symm(const Layout layout, const Side side, const Triangle triangle,
                 const size_t m, const size_t n,
diff --git a/include/internal/database.h b/include/internal/database.h
index dbbdd5c0..33ad1979 100644
--- a/include/internal/database.h
+++ b/include/internal/database.h
@@ -54,6 +54,7 @@ class Database {
 
   // The database consists of separate database entries, stored together in a vector
   static const DatabaseEntry XaxpySingle, XaxpyDouble, XaxpyComplexSingle, XaxpyComplexDouble;
+  static const DatabaseEntry XgemvSingle, XgemvDouble, XgemvComplexSingle, XgemvComplexDouble;
   static const DatabaseEntry XgemmSingle, XgemmDouble, XgemmComplexSingle, XgemmComplexDouble;
   static const DatabaseEntry CopySingle, CopyDouble, CopyComplexSingle, CopyComplexDouble;
   static const DatabaseEntry PadSingle, PadDouble, PadComplexSingle, PadComplexDouble;
diff --git a/include/internal/database/xgemv.h b/include/internal/database/xgemv.h
new file mode 100644
index 00000000..ef45f486
--- /dev/null
+++ b/include/internal/database/xgemv.h
@@ -0,0 +1,129 @@
+
+// =================================================================================================
+// This file is part of the CLBlast project. The project is licensed under Apache Version 2.0. This
+// project loosely follows the Google C++ styleguide and uses a tab-size of two spaces and a max-
+// width of 100 characters per line.
+//
+// Author(s):
+//   Cedric Nugteren <www.cedricnugteren.nl>
+//
+// This file populates the database with best-found tuning parameters for the Xgemv kernels.
+//
+// =================================================================================================
+
+namespace clblast {
+// =================================================================================================
+
+const Database::DatabaseEntry Database::XgemvSingle = {
+  "Xgemv", Precision::kSingle, {
+    { // NVIDIA GPUs
+      CL_DEVICE_TYPE_GPU, "NVIDIA Corporation", {
+        { "GeForce GTX 480",  { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+        { "Tesla K20m",       { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+        { "Tesla K40m",       { {"WGS1",256}, {"WPT1",1}, {"WGS2",256}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",4} } },
+      }
+    },
+    { // AMD GPUs
+      CL_DEVICE_TYPE_GPU, "AMD", {
+        { "Tahiti",           { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+      }
+    },
+    { // Intel GPUs
+      CL_DEVICE_TYPE_GPU, "Intel", {
+        { "Iris",             { {"WGS1",256}, {"WPT1",2}, {"WGS2",64}, {"WPT2",4}, {"VW2",4}, {"WGS3",256}, {"WPT3",2}, {"VW3",8} } },
+      }
+    },
+    { // Default
+      CL_DEVICE_TYPE_ALL, kDefault, {
+        { kDefault,           { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+      }
+    },
+  }
+};
+
+// =================================================================================================
+
+const Database::DatabaseEntry Database::XgemvDouble = {
+  "Xgemv", Precision::kDouble, {
+    { // NVIDIA GPUs
+      CL_DEVICE_TYPE_GPU, "NVIDIA Corporation", {
+        { "GeForce GTX 480",  { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+        { "Tesla K20m",       { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+        { "Tesla K40m",       { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+      }
+    },
+    { // AMD GPUs
+      CL_DEVICE_TYPE_GPU, "AMD", {
+        { "Tahiti",           { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+      }
+    },
+    { // Intel GPUs
+      CL_DEVICE_TYPE_GPU, "Intel", {
+      }
+    },
+    { // Default
+      CL_DEVICE_TYPE_ALL, kDefault, {
+        { kDefault,           { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+      }
+    },
+  }
+};
+// =================================================================================================
+
+const Database::DatabaseEntry Database::XgemvComplexSingle = {
+  "Xgemv", Precision::kComplexSingle, {
+    { // NVIDIA GPUs
+      CL_DEVICE_TYPE_GPU, "NVIDIA Corporation", {
+        { "GeForce GTX 480",  { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+        { "Tesla K20m",       { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+        { "Tesla K40m",       { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+      }
+    },
+    { // AMD GPUs
+      CL_DEVICE_TYPE_GPU, "AMD", {
+        { "Tahiti",           { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+      }
+    },
+    { // Intel GPUs
+      CL_DEVICE_TYPE_GPU, "Intel", {
+        { "Iris",             { {"WGS1",256}, {"WPT1",1}, {"WGS2",64}, {"WPT2",4}, {"VW2",2}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+      }
+    },
+    { // Default
+      CL_DEVICE_TYPE_ALL, kDefault, {
+        { kDefault,           { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+      }
+    },
+  }
+};
+
+// =================================================================================================
+
+const Database::DatabaseEntry Database::XgemvComplexDouble = {
+  "Xgemv", Precision::kComplexDouble, {
+    { // NVIDIA GPUs
+      CL_DEVICE_TYPE_GPU, "NVIDIA Corporation", {
+        { "GeForce GTX 480",  { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+        { "Tesla K20m",       { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+        { "Tesla K40m",       { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+      }
+    },
+    { // AMD GPUs
+      CL_DEVICE_TYPE_GPU, "AMD", {
+        { "Tahiti",           { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+      }
+    },
+    { // Intel GPUs
+      CL_DEVICE_TYPE_GPU, "Intel", {
+      }
+    },
+    { // Default
+      CL_DEVICE_TYPE_ALL, kDefault, {
+        { kDefault,           { {"WGS1",64}, {"WPT1",1}, {"WGS2",64}, {"WPT2",1}, {"VW2",1}, {"WGS3",64}, {"WPT3",1}, {"VW3",1} } },
+      }
+    },
+  }
+};
+
+// =================================================================================================
+} // namespace clblast
diff --git a/include/internal/routines/xgemv.h b/include/internal/routines/xgemv.h
new file mode 100644
index 00000000..a3109036
--- /dev/null
+++ b/include/internal/routines/xgemv.h
@@ -0,0 +1,46 @@
+
+// =================================================================================================
+// This file is part of the CLBlast project. The project is licensed under Apache Version 2.0. This
+// project loosely follows the Google C++ styleguide and uses a tab-size of two spaces and a max-
+// width of 100 characters per line.
+//
+// Author(s):
+//   Cedric Nugteren <www.cedricnugteren.nl>
+//
+// This file implements the Xgemv routine. The precision is implemented using a template argument.
+//
+// =================================================================================================
+
+#ifndef CLBLAST_ROUTINES_XGEMV_H_
+#define CLBLAST_ROUTINES_XGEMV_H_
+
+#include "internal/routine.h"
+
+namespace clblast {
+// =================================================================================================
+
+// See comment at top of file for a description of the class
+template <typename T>
+class Xgemv: public Routine {
+ public:
+  Xgemv(CommandQueue &queue, Event &event);
+
+  // Templated-precision implementation of the routine
+  StatusCode DoGemv(const Layout layout, const Transpose a_transpose,
+                    const size_t m, const size_t n,
+                    const T alpha,
+                    const Buffer &a_buffer, const size_t a_offset, const size_t a_ld,
+                    const Buffer &x_buffer, const size_t x_offset, const size_t x_inc,
+                    const T beta,
+                    const Buffer &y_buffer, const size_t y_offset, const size_t y_inc);
+
+ private:
+  // Static variable to get the precision
+  const static Precision precision_;
+};
+
+// =================================================================================================
+} // namespace clblast
+
+// CLBLAST_ROUTINES_XGEMV_H_
+#endif
diff --git a/include/internal/tuning.h b/include/internal/tuning.h
index 7768888c..d0cf6b5d 100644
--- a/include/internal/tuning.h
+++ b/include/internal/tuning.h
@@ -34,10 +34,20 @@ using Tuner3 = std::function<void(const Arguments<T>&,
                                   const std::vector<T>&, const std::vector<T>&, std::vector<T>&,
                                   cltune::Tuner&)>;
 
+// As above, but now with an additional ID for the variation
+template <typename T>
+using Tuner3V = std::function<void(const Arguments<T>&, const size_t,
+                                   const std::vector<T>&, const std::vector<T>&, std::vector<T>&,
+                                   cltune::Tuner&)>;
+
 // Tuner for vector-vector input
 template <typename T>
 void TunerXY(int argc, char* argv[], const Tuner2<T> &tune_function);
 
+// Tuner for matrix-vector-vector input
+template <typename T>
+void TunerAXY(int argc, char* argv[], const size_t num_variations, const Tuner3V<T> &tune_function);
+
 // Tuner for matrix-matrix input
 template <typename T>
 void TunerAB(int argc, char* argv[], const Tuner2<T> &tune_function);
author	Cedric Nugteren <web@cedricnugteren.nl>	2015-06-15 18:35:27 +0200
committer	Cedric Nugteren <web@cedricnugteren.nl>	2015-06-15 18:35:27 +0200
commit	9e2fba9ab9cab1f94dfe143fc6e163f47b6d6f39 (patch)
tree	f4e556ee5bda4e634c7af1db094e11f6def4e07e /include
parent	3c17c1c13313022879c8caf289d0f47ea5d7d22d (diff)
parent	f925d47dadb5f7dd161e44e5da2142734faa3176 (diff)