diff options
author | Cedric Nugteren <web@cedricnugteren.nl> | 2016-05-16 12:37:24 +0200 |
---|---|---|
committer | Cedric Nugteren <web@cedricnugteren.nl> | 2016-05-16 12:37:24 +0200 |
commit | af2ac6221288ec101a69018e960bb004ad698efe (patch) | |
tree | 9173d204820075598dd2fe13a0cb1771fd732945 /src/routines/level3/xgemm.cc | |
parent | 591e343ec94077f873b1aa12052a4ce55ae80200 (diff) |
Prepared GEMM and supporting kernels and tuners for half-precision support
Diffstat (limited to 'src/routines/level3/xgemm.cc')
-rw-r--r-- | src/routines/level3/xgemm.cc | 11 |
1 files changed, 9 insertions, 2 deletions
diff --git a/src/routines/level3/xgemm.cc b/src/routines/level3/xgemm.cc index 11116aae..5395667a 100644 --- a/src/routines/level3/xgemm.cc +++ b/src/routines/level3/xgemm.cc @@ -123,6 +123,12 @@ StatusCode Xgemm<T>::DoGemm(const Layout layout, auto b_temp = (b_no_temp) ? b_buffer : Buffer<T>(context_, k_ceiled*n_ceiled); auto c_temp = (c_no_temp) ? c_buffer : Buffer<T>(context_, m_ceiled*n_ceiled); + // Upload the scalar arguments as constant buffers to the device (needed for half-precision) + auto alpha_buffer = Buffer<T>(context_, 1); + auto beta_buffer = Buffer<T>(context_, 1); + alpha_buffer.Write(queue_, 1, &alpha); + beta_buffer.Write(queue_, 1, &beta); + // Events of all kernels (including pre/post processing kernels) auto eventWaitList = std::vector<Event>(); auto emptyEventList = std::vector<Event>(); @@ -170,8 +176,8 @@ StatusCode Xgemm<T>::DoGemm(const Layout layout, kernel.SetArgument(0, static_cast<int>(m_ceiled)); kernel.SetArgument(1, static_cast<int>(n_ceiled)); kernel.SetArgument(2, static_cast<int>(k_ceiled)); - kernel.SetArgument(3, alpha); - kernel.SetArgument(4, beta); + kernel.SetArgument(3, alpha_buffer()); + kernel.SetArgument(4, beta_buffer()); kernel.SetArgument(5, a_temp()); kernel.SetArgument(6, b_temp()); kernel.SetArgument(7, c_temp()); @@ -207,6 +213,7 @@ StatusCode Xgemm<T>::DoGemm(const Layout layout, // ================================================================================================= // Compiles the templated class +template class Xgemm<half>; template class Xgemm<float>; template class Xgemm<double>; template class Xgemm<float2>; |