diff options
author | CNugteren <web@cedricnugteren.nl> | 2015-06-24 07:50:18 +0200 |
---|---|---|
committer | CNugteren <web@cedricnugteren.nl> | 2015-06-24 07:50:18 +0200 |
commit | 60a88aac8672d360eb05ba25b1c4ffbf53a78dff (patch) | |
tree | 25b6c8d59b293b3c7e0d7fb48bb8b6ca64a1f2d9 /include | |
parent | a17297937d757d9747adde600f832d1e0c2753c1 (diff) |
Added the SYRK routine, tester, and client
Diffstat (limited to 'include')
-rw-r--r-- | include/internal/routines/xsyrk.h | 49 |
1 files changed, 49 insertions, 0 deletions
diff --git a/include/internal/routines/xsyrk.h b/include/internal/routines/xsyrk.h new file mode 100644 index 00000000..3dab731f --- /dev/null +++ b/include/internal/routines/xsyrk.h @@ -0,0 +1,49 @@ + +// ================================================================================================= +// This file is part of the CLBlast project. The project is licensed under Apache Version 2.0. This +// project loosely follows the Google C++ styleguide and uses a tab-size of two spaces and a max- +// width of 100 characters per line. +// +// Author(s): +// Cedric Nugteren <www.cedricnugteren.nl> +// +// This file implements the Xsyrk routine. The precision is implemented using a template argument. +// The implementation is based on the regular Xgemm routine and kernel, but with two main changes: +// 1) The final unpad(transpose) kernel updates only the upper/lower triangular part. +// 2) The main Xgemm kernel masks workgroups not contributing to usefull data. This is only for +// performance reasons, as the actual masking is done later (see the first point). +// +// ================================================================================================= + +#ifndef CLBLAST_ROUTINES_XSYRK_H_ +#define CLBLAST_ROUTINES_XSYRK_H_ + +#include "internal/routine.h" + +namespace clblast { +// ================================================================================================= + +// See comment at top of file for a description of the class +template <typename T> +class Xsyrk: public Routine { + public: + Xsyrk(CommandQueue &queue, Event &event); + + // Templated-precision implementation of the routine + StatusCode DoSyrk(const Layout layout, const Triangle triangle, const Transpose a_transpose, + const size_t n, const size_t k, + const T alpha, + const Buffer &a_buffer, const size_t a_offset, const size_t a_ld, + const T beta, + const Buffer &c_buffer, const size_t c_offset, const size_t c_ld); + + private: + // Static variable to get the precision + const static Precision precision_; +}; + +// ================================================================================================= +} // namespace clblast + +// CLBLAST_ROUTINES_XSYRK_H_ +#endif |