Merge pull request #99 from CNugteren/development

Update to version 0.9.0
author: Cedric Nugteren <web@cedricnugteren.nl> 2016-09-13 21:14:51 +0200
committer: GitHub <noreply@github.com> 2016-09-13 21:14:51 +0200
commit: f07ac22f5b57d22756d779d2e53620f988d786ee (patch)
tree: e8bcbc331683ca6fd807f5a5b83bb05c6e6fed69 /src/kernels/level3/copy_pad.opencl
parent: 7c13bacf129291e3e295ecb6e833788477085fa0 (diff)
parent: 4b94afda941a86f363064ff02f97e21eb9618794 (diff)
1 files changed, 21 insertions, 21 deletions
diff --git a/src/kernels/level3/copy_pad.opencl b/src/kernels/level3/copy_pad.opencl
index d276cc60..29480b25 100644
--- a/src/kernels/level3/copy_pad.opencl
+++ b/src/kernels/level3/copy_pad.opencl
@@ -24,16 +24,16 @@ R"(
 // Copies a matrix from source to destination. The output is padded with zero values in case the
 // destination matrix dimensions are larger than the source matrix dimensions. Additionally, the ld
 // value and offset can be different.
-__attribute__((reqd_work_group_size(PAD_DIMX, PAD_DIMY, 1)))
-__kernel void CopyPadMatrix(const int src_one, const int src_two,
-                            const int src_ld, const int src_offset,
-                            __global const real* restrict src,
-                            const int dest_one, const int dest_two,
-                            const int dest_ld, const int dest_offset,
-                            __global real* dest,
-                            const __constant real* restrict arg_alpha,
-                            const int do_conjugate) {
-  const real alpha = arg_alpha[0];
+__kernel __attribute__((reqd_work_group_size(PAD_DIMX, PAD_DIMY, 1)))
+void CopyPadMatrix(const int src_one, const int src_two,
+                   const int src_ld, const int src_offset,
+                   __global const real* restrict src,
+                   const int dest_one, const int dest_two,
+                   const int dest_ld, const int dest_offset,
+                   __global real* dest,
+                   const real_arg arg_alpha,
+                   const int do_conjugate) {
+  const real alpha = GetRealArg(arg_alpha);
 
   // Loops over the work per thread in both dimensions
   #pragma unroll
@@ -65,17 +65,17 @@ __kernel void CopyPadMatrix(const int src_one, const int src_two,
 // Same as above, but now un-pads a matrix. This kernel reads data from a padded source matrix, but
 // writes only the actual data back to the destination matrix. Again, the ld value and offset can
 // be different.
-__attribute__((reqd_work_group_size(PAD_DIMX, PAD_DIMY, 1)))
-__kernel void CopyMatrix(const int src_one, const int src_two,
-                         const int src_ld, const int src_offset,
-                         __global const real* restrict src,
-                         const int dest_one, const int dest_two,
-                         const int dest_ld, const int dest_offset,
-                         __global real* dest,
-                         const __constant real* restrict arg_alpha,
-                         const int upper, const int lower,
-                         const int diagonal_imag_zero) {
-  const real alpha = arg_alpha[0];
+__kernel __attribute__((reqd_work_group_size(PAD_DIMX, PAD_DIMY, 1)))
+void CopyMatrix(const int src_one, const int src_two,
+                const int src_ld, const int src_offset,
+                __global const real* restrict src,
+                const int dest_one, const int dest_two,
+                const int dest_ld, const int dest_offset,
+                __global real* dest,
+                const real_arg arg_alpha,
+                const int upper, const int lower,
+                const int diagonal_imag_zero) {
+  const real alpha = GetRealArg(arg_alpha);
 
   // Loops over the work per thread in both dimensions
   #pragma unroll
author	Cedric Nugteren <web@cedricnugteren.nl>	2016-09-13 21:14:51 +0200
committer	GitHub <noreply@github.com>	2016-09-13 21:14:51 +0200
commit	f07ac22f5b57d22756d779d2e53620f988d786ee (patch)
tree	e8bcbc331683ca6fd807f5a5b83bb05c6e6fed69 /src/kernels/level3/copy_pad.opencl
parent	7c13bacf129291e3e295ecb6e833788477085fa0 (diff)
parent	4b94afda941a86f363064ff02f97e21eb9618794 (diff)