ROCm
diff --git a/‎example/09_convnd_fwd/convnd_fwd_common.hpp‎
Lines changed: 62 additions & 3 deletions b/‎example/09_convnd_fwd/convnd_fwd_common.hpp‎
Lines changed: 62 additions & 3 deletions
diff --git a/‎example/09_convnd_fwd/convnd_fwd_dl_common.hpp‎
Lines changed: 2 additions & 1 deletion b/‎example/09_convnd_fwd/convnd_fwd_dl_common.hpp‎
Lines changed: 2 additions & 1 deletion
diff --git a/‎example/09_convnd_fwd/run_convnd_fwd_example.inc‎
Lines changed: 3 additions & 3 deletions b/‎example/09_convnd_fwd/run_convnd_fwd_example.inc‎
Lines changed: 3 additions & 3 deletions
diff --git a/‎example/17_convnd_bwd_data/convnd_bwd_data_common.hpp‎
Lines changed: 103 additions & 5 deletions b/‎example/17_convnd_bwd_data/convnd_bwd_data_common.hpp‎
Lines changed: 103 additions & 5 deletions
diff --git a/‎example/17_convnd_bwd_data/convnd_bwd_data_xdl_fp16.cpp‎
Lines changed: 3 additions & 3 deletions b/‎example/17_convnd_bwd_data/convnd_bwd_data_xdl_fp16.cpp‎
Lines changed: 3 additions & 3 deletions
diff --git a/‎example/20_grouped_conv_bwd_weight/common.hpp‎
Lines changed: 46 additions & 3 deletions b/‎example/20_grouped_conv_bwd_weight/common.hpp‎
Lines changed: 46 additions & 3 deletions
@@ -18,14 +18,16 @@
 #include "ck/library/utility/convolution_parameter.hpp"
 #include "ck/library/utility/convolution_host_tensor_descriptor_helper.hpp"
 #include "ck/library/reference_tensor_operation/cpu/reference_conv_fwd.hpp"
+#include "ck/library/reference_tensor_operation/gpu/naive_conv_fwd_gpu.hpp"
+#include "ck_tile/host/hip_check_error.hpp"
 
 using ::ck::DeviceMem;
 using ::ck::HostTensorDescriptor;
 using ::ck::Tensor;
 
 void print_helper_msg()
 {
-    std::cout << "arg1: verification (0=no, 1=yes)\n"
+    std::cout << "arg1: verification (0=no, 1=CPU, 2=GPU)\n"
               << "arg2: initialization (0=no init, 1=integer value, 2=decimal value)\n"
               << "arg3: time kernel (0=no, 1=yes)\n"
               << ck::utils::conv::get_conv_param_parser_helper_msg() << std::endl;
@@ -130,7 +132,7 @@ template <ck::index_t NDimSpatial,
           typename OutElementOp,
           typename DeviceConvNDFwdInstance,
           typename ComputeDataType = OutDataType>
-bool run_grouped_conv_fwd(bool do_verification,
+bool run_grouped_conv_fwd(int do_verification,
                           int init_method,
                           bool time_kernel,
                           const ck::utils::conv::ConvParam& conv_param,
@@ -233,8 +235,11 @@ bool run_grouped_conv_fwd(bool do_verification,
     std::cout << "Perf: " << avg_time << " ms, " << tflops << " TFlops, " << gb_per_sec << " GB/s, "
               << conv.GetTypeString() << std::endl;
 
-    if(do_verification)
+    std::cout << "do_verification = " << do_verification << std::endl;
+
+    if(do_verification == 1)
     {
+        // CPU verification
         auto ref_conv = ck::tensor_operation::host::ReferenceConvFwd<NDimSpatial,
                                                                      InDataType,
                                                                      WeiDataType,
@@ -269,6 +274,60 @@ bool run_grouped_conv_fwd(bool do_verification,
                                     get_rtol<OutDataType, ComputeDataType>(),
                                     get_atol<OutDataType, ComputeDataType>());
     }
+    else if(do_verification == 2)
+    {
+        // GPU verification using naive GPU reference
+        std::cout << "Running GPU verification..." << std::endl;
+
+        // Allocate and ZERO GPU memory for reference output
+        DeviceMem out_device_ref_buf(sizeof(OutDataType) * out_device.mDesc.GetElementSpaceSize());
+        out_device_ref_buf.SetZero();
+
+        // Extract dimensions using helper function
+        ck::ref::ConvDims dims = ck::utils::conv::extract_conv_dims(conv_param, NDimSpatial);
+
+        // Launch GPU reference kernel
+        constexpr ck::index_t block_size     = 256;
+        const ck::long_index_t output_length = dims.N * dims.Do * dims.Ho * dims.Wo * dims.K;
+        const ck::index_t grid_size          = (output_length + block_size - 1) / block_size;
+
+        auto gpu_ref_kernel = ck::ref::naive_conv_fwd_ndhwc_kzyxc_ndhwk<InDataType,
+                                                                        WeiDataType,
+                                                                        OutDataType,
+                                                                        ComputeDataType,
+                                                                        InElementOp,
+                                                                        WeiElementOp,
+                                                                        OutElementOp>;
+
+        gpu_ref_kernel<<<dim3(grid_size), dim3(block_size), 0, nullptr>>>(
+            reinterpret_cast<const InDataType*>(in_device_buf.GetDeviceBuffer()),
+            reinterpret_cast<const WeiDataType*>(wei_device_buf.GetDeviceBuffer()),
+            reinterpret_cast<OutDataType*>(out_device_ref_buf.GetDeviceBuffer()),
+            dims);
+
+        HIP_CHECK_ERROR(hipDeviceSynchronize());
+
+        std::cout << "GPU reference kernel completed successfully, copying results..." << std::endl;
+
+        // Copy GPU reference result to host
+        out_device_ref_buf.FromDevice(out_host.mData.data());
+
+        // Copy GPU kernel result to host
+        out_device_buf.FromDevice(out_device.mData.data());
+
+        std::cout << "Comparing GPU kernel output vs GPU reference..." << std::endl;
+
+        // Compare GPU kernel vs GPU reference
+        bool pass = ck::utils::check_err(out_device,
+                                         out_host,
+                                         "Error: incorrect results!",
+                                         get_rtol<OutDataType, ComputeDataType>(),
+                                         get_atol<OutDataType, ComputeDataType>());
+
+        std::cout << "GPU verification result is:" << (pass ? "correct" : "fail") << std::endl;
+
+        return pass;
+    }
 
     return true;
 }
@@ -25,7 +25,7 @@ using ::ck::Tensor;
 
 void print_helper_msg()
 {
-    std::cout << "arg1: verification (0=no, 1=yes)\n"
+    std::cout << "arg1: verification (0=no, 1=CPU)\n"
               << "arg2: initialization (0=no init, 1=integer value, 2=decimal value)\n"
               << "arg3: time kernel (0=no, 1=yes)\n"
               << ck::utils::conv::get_conv_param_parser_helper_msg() << std::endl;
@@ -162,6 +162,7 @@ bool run_grouped_conv_fwd_dl(bool do_verification,
 
     if(do_verification)
     {
+        // CPU verification only (DL variants are fused operations)
         auto ref_conv = ck::tensor_operation::host::ReferenceConvFwd<
             NDimSpatial,
             InDataType,
 
@@ -12,9 +12,9 @@ bool run_convnd_fwd_example(int argc, char* argv[])
 {
     print_helper_msg();
 
-    bool do_verification = true;
-    int init_method      = 1;
-    bool time_kernel     = false;
+    int do_verification = 1; // 0=no, 1=CPU, 2=GPU
+    int init_method     = 1;
+    bool time_kernel    = false;
 
     ck::utils::conv::ConvParam conv_param{
         2, 1, 128, 256, 192, {3, 3}, {71, 71}, {2, 2}, {1, 1}, {1, 1}, {1, 1}};
 
@@ -17,14 +17,58 @@
 #include "ck/library/utility/convolution_parameter.hpp"
 #include "ck/library/utility/convolution_host_tensor_descriptor_helper.hpp"
 #include "ck/library/reference_tensor_operation/cpu/reference_conv_bwd_data.hpp"
+#include "ck/library/reference_tensor_operation/gpu/naive_conv_bwd_data_gpu.hpp"
+#include "ck_tile/host/hip_check_error.hpp"
 
 using ::ck::DeviceMem;
 using ::ck::HostTensorDescriptor;
 using ::ck::Tensor;
 
+template <typename DataType, typename GemmType = DataType>
+inline __host__ __device__ constexpr double get_rtol()
+{
+    if constexpr(std::is_same_v<DataType, float> && std::is_same_v<GemmType, ck::tf32_t>)
+        return 5e-3;
+    else if constexpr(std::is_same_v<DataType, float>)
+        return 1e-3;
+    else if constexpr(std::is_same_v<DataType, double>)
+        return 1e-6;
+    else if constexpr(std::is_same_v<DataType, ck::half_t>)
+        return 1e-3;
+    else if constexpr(std::is_same_v<DataType, ck::bhalf_t>)
+        return 5e-2;
+    else if constexpr(std::is_same_v<DataType, ck::f8_t>)
+        return 1e-1;
+    else if constexpr(std::is_same_v<DataType, ck::bf8_t>)
+        return 1.5e-1;
+    else
+        return 1e-3;
+}
+
+template <typename DataType, typename GemmType = DataType>
+inline __host__ __device__ constexpr double get_atol()
+{
+    if constexpr(std::is_same_v<DataType, float> && std::is_same_v<GemmType, ck::tf32_t>)
+        return 1e-3;
+    else if constexpr(std::is_same_v<DataType, float>)
+        return 1e-3;
+    else if constexpr(std::is_same_v<DataType, double>)
+        return 1e-6;
+    else if constexpr(std::is_same_v<DataType, ck::half_t>)
+        return 1e-3;
+    else if constexpr(std::is_same_v<DataType, ck::bhalf_t>)
+        return 5e-2;
+    else if constexpr(std::is_same_v<DataType, ck::f8_t>)
+        return 16.1;
+    else if constexpr(std::is_same_v<DataType, ck::bf8_t>)
+        return 16.1;
+    else
+        return 1e-3;
+}
+
 void print_helper_msg()
 {
-    std::cout << "arg1: verification (0=no, 1=yes)\n"
+    std::cout << "arg1: verification (0=no, 1=CPU, 2=GPU)\n"
               << "arg2: initialization (0=no init, 1=integer value, 2=decimal value)\n"
               << "arg3: time kernel (0=no, 1=yes)\n"
               << ck::utils::conv::get_conv_param_parser_helper_msg() << std::endl;
@@ -38,7 +82,7 @@ template <ck::index_t NDimSpatial,
           typename WeiElementOp,
           typename OutElementOp,
           typename DeviceConvNdBwdDataInstance>
-int run_conv_bwd_data(bool do_verification,
+int run_conv_bwd_data(int do_verification,
                       int init_method,
                       bool time_kernel,
                       const ck::utils::conv::ConvParam& conv_param,
@@ -128,26 +172,30 @@ int run_conv_bwd_data(bool do_verification,
                                  wei_element_op,
                                  out_element_op);
 
+    // Check if optimized kernel supports these parameters
     if(!conv.IsSupportedArgument(argument.get()))
     {
         std::cout << "Not support,please check parameters or device";
         return 0;
     }
 
+    // Run optimized kernel
     float ave_time = invoker.Run(argument.get(), StreamConfig{nullptr, time_kernel});
 
     std::size_t flop      = conv_param.GetFlops();
     std::size_t num_btype = conv_param.GetByte<InDataType, WeiDataType, OutDataType>();
 
-    float tflops = static_cast<float>(flop) / 1.E9 / ave_time;
-
+    float tflops     = static_cast<float>(flop) / 1.E9 / ave_time;
     float gb_per_sec = num_btype / 1.E6 / ave_time;
 
     std::cout << "Perf: " << ave_time << " ms, " << tflops << " TFlops, " << gb_per_sec << " GB/s"
               << std::endl;
 
-    if(do_verification)
+    std::cout << "do_verification = " << do_verification << std::endl;
+
+    if(do_verification == 1)
     {
+        // CPU verification
         auto ref_conv = ck::tensor_operation::host::ReferenceConvBwdData<NDimSpatial,
                                                                          InDataType,
                                                                          WeiDataType,
@@ -175,6 +223,56 @@ int run_conv_bwd_data(bool do_verification,
 
         return ck::utils::check_err(in_device, in_host) ? 0 : 1;
     }
+    else if(do_verification == 2)
+    {
+        // GPU verification
+        std::cout << "Running GPU verification..." << std::endl;
+
+        DeviceMem in_device_ref_buf(sizeof(InDataType) * in_device.mDesc.GetElementSpaceSize());
+        in_device_ref_buf.SetZero();
+
+        // Extract dimensions using helper function
+        ck::ref::ConvDims dims = ck::utils::conv::extract_conv_dims(conv_param, NDimSpatial);
+
+        constexpr ck::index_t block_size    = 256;
+        const ck::long_index_t input_length = dims.N * dims.Di * dims.Hi * dims.Wi * dims.C;
+        const ck::index_t grid_size         = (input_length + block_size - 1) / block_size;
+
+        auto gpu_ref_kernel = ck::ref::naive_conv_bwd_data_ndhwc_kzyxc_ndhwk<InDataType,
+                                                                             WeiDataType,
+                                                                             OutDataType,
+                                                                             float,
+                                                                             InElementOp,
+                                                                             WeiElementOp,
+                                                                             OutElementOp>;
+
+        gpu_ref_kernel<<<dim3(grid_size), dim3(block_size), 0, nullptr>>>(
+            reinterpret_cast<InDataType*>(in_device_ref_buf.GetDeviceBuffer()),
+            reinterpret_cast<const WeiDataType*>(wei_device_buf.GetDeviceBuffer()),
+            reinterpret_cast<const OutDataType*>(out_device_buf.GetDeviceBuffer()),
+            dims);
+
+        HIP_CHECK_ERROR(hipDeviceSynchronize());
+
+        std::cout << "GPU reference kernel completed, copying results..." << std::endl;
+
+        // Copy GPU reference result
+        Tensor<InDataType> in_gpu_ref(in_host.mDesc);
+        in_device_ref_buf.FromDevice(in_gpu_ref.mData.data());
+
+        // Copy optimized kernel result
+        in_device_buf.FromDevice(in_device.mData.data());
+
+        // Compare: Optimized kernel result vs GPU reference result
+        bool pass = ck::utils::check_err(in_device,
+                                         in_gpu_ref,
+                                         "Error: Incorrect results!",
+                                         get_rtol<InDataType, float>(),
+                                         get_atol<InDataType, float>());
+        std::cout << "GPU verification result is:" << (pass ? "correct" : "fail") << std::endl;
+
+        return pass ? 0 : 1;
+    }
 
     return 0;
 }
@@ -63,9 +63,9 @@ int main(int argc, char* argv[])
 
     print_helper_msg();
 
-    bool do_verification = true;
-    int init_method      = 1;
-    bool time_kernel     = false;
+    int do_verification = 1; // 0=no, 1=CPU, 2=GPU
+    int init_method     = 1;
+    bool time_kernel    = false;
 
     ck::utils::conv::ConvParam conv_param{
         2, 1, 128, 256, 256, {3, 3}, {71, 71}, {2, 2}, {1, 1}, {1, 1}, {1, 1}};
 
@@ -19,6 +19,7 @@
 #include "ck/library/utility/convolution_parameter.hpp"
 #include "ck/library/utility/convolution_host_tensor_descriptor_helper.hpp"
 #include "ck/library/reference_tensor_operation/cpu/reference_conv_bwd_weight.hpp"
+#include "ck/library/reference_tensor_operation/gpu/naive_conv_bwd_weight_gpu.hpp"
 
 using ::ck::DeviceMem;
 using ::ck::HostTensorDescriptor;
@@ -38,6 +39,48 @@ using PassThrough = ck::tensor_operation::element_wise::PassThrough;
 static constexpr auto ConvBwdWeightDefault =
     ck::tensor_operation::device::ConvolutionBackwardWeightSpecialization::Default;
 
+template <typename DataType, typename GemmType = DataType>
+inline __host__ __device__ constexpr double get_rtol()
+{
+    if constexpr(std::is_same_v<DataType, float> && std::is_same_v<GemmType, ck::tf32_t>)
+        return 5e-3;
+    else if constexpr(std::is_same_v<DataType, float>)
+        return 1e-3;
+    else if constexpr(std::is_same_v<DataType, double>)
+        return 1e-6;
+    else if constexpr(std::is_same_v<DataType, ck::half_t>)
+        return 1e-3;
+    else if constexpr(std::is_same_v<DataType, ck::bhalf_t>)
+        return 5e-2;
+    else if constexpr(std::is_same_v<DataType, ck::f8_t>)
+        return 1e-1;
+    else if constexpr(std::is_same_v<DataType, ck::bf8_t>)
+        return 1.5e-1;
+    else
+        return 1e-3;
+}
+
+template <typename DataType, typename GemmType = DataType>
+inline __host__ __device__ constexpr double get_atol()
+{
+    if constexpr(std::is_same_v<DataType, float> && std::is_same_v<GemmType, ck::tf32_t>)
+        return 1e-3;
+    else if constexpr(std::is_same_v<DataType, float>)
+        return 1e-3;
+    else if constexpr(std::is_same_v<DataType, double>)
+        return 1e-6;
+    else if constexpr(std::is_same_v<DataType, ck::half_t>)
+        return 1e-3;
+    else if constexpr(std::is_same_v<DataType, ck::bhalf_t>)
+        return 5e-2;
+    else if constexpr(std::is_same_v<DataType, ck::f8_t>)
+        return 16.1;
+    else if constexpr(std::is_same_v<DataType, ck::bf8_t>)
+        return 16.1;
+    else
+        return 1e-3;
+}
+
 template <typename InputLay, typename WeightLay, typename OutputLay>
 struct CommonLayoutSetting
 {
@@ -75,9 +118,9 @@ using OutputLayout = typename CommonLayoutSettingSelector<NDimSpatial>::OutputLa
 
 struct ExecutionConfig final
 {
-    bool do_verification = true;
-    int init_method      = 1;
-    bool time_kernel     = false;
+    int do_verification = 1; // 0=no, 1=CPU, 2=GPU
+    int init_method     = 1;
+    bool time_kernel    = false;
 };
 
 #define DefaultConvParam                                                                         \
Original file line number	Diff line number	Diff line change
`@@ -25,7 +25,7 @@ using ::ck::Tensor;`
`25`	`25`
`26`	`26`	`void print_helper_msg()`
`27`	`27`	`{`
`28`		`- std::cout << "arg1: verification (0=no, 1=yes)\n"`
	`28`	`+ std::cout << "arg1: verification (0=no, 1=CPU)\n"`
`29`	`29`	`<< "arg2: initialization (0=no init, 1=integer value, 2=decimal value)\n"`
`30`	`30`	`<< "arg3: time kernel (0=no, 1=yes)\n"`
`31`	`31`	`<< ck::utils::conv::get_conv_param_parser_helper_msg() << std::endl;`
`@@ -162,6 +162,7 @@ bool run_grouped_conv_fwd_dl(bool do_verification,`
`162`	`162`
`163`	`163`	`if(do_verification)`
`164`	`164`	`{`
	`165`	`+ // CPU verification only (DL variants are fused operations)`
`165`	`166`	`auto ref_conv = ck::tensor_operation::host::ReferenceConvFwd<`
`166`	`167`	`NDimSpatial,`
`167`	`168`	`InDataType,`