gemm_util.hpp 10.2 KB
Newer Older
Chao Liu's avatar
Chao Liu committed
1
2
3
// SPDX-License-Identifier: MIT
// Copyright (c) 2018-2022, Advanced Micro Devices, Inc. All rights reserved.

Chao Liu's avatar
Chao Liu committed
4
#pragma once
Anthony Chang's avatar
Anthony Chang committed
5

Chao Liu's avatar
Chao Liu committed
6
7
8
#include "ck/ck.hpp"
#include "ck/tensor_operation/gpu/device/tensor_layout.hpp"
#include "ck/library/utility/check_err.hpp"
9
10
11
#include "ck/library/utility/device_memory.hpp"
#include "ck/library/utility/host_tensor.hpp"
#include "ck/library/utility/host_tensor_generator.hpp"
12
#include "ck/library/utility/literals.hpp"
Chao Liu's avatar
Chao Liu committed
13
#include "ck/library/reference_tensor_operation/cpu/reference_gemm.hpp"
Anthony Chang's avatar
Anthony Chang committed
14
15
16
17
18
19

namespace ck {
namespace gemm_util {

struct GemmParams
{
20
21
22
    ck::index_t M = 1024;
    ck::index_t N = 1024;
    ck::index_t K = 1024;
Anthony Chang's avatar
Anthony Chang committed
23

24
25
26
    ck::index_t StrideA = 1024;
    ck::index_t StrideB = 1024;
    ck::index_t StrideC = 1024;
Anthony Chang's avatar
Anthony Chang committed
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
};

template <typename GemmInstance,
          typename ADataType,
          typename BDataType,
          typename CDataType,
          typename AElementwiseOperation,
          typename BElementwiseOperation,
          typename CElementwiseOperation>
void RunHostGEMM(const Tensor<ADataType>& A,
                 const Tensor<BDataType>& B,
                 Tensor<CDataType>& C,
                 AElementwiseOperation a_element_op,
                 BElementwiseOperation b_element_op,
                 CElementwiseOperation c_element_op)
{
    auto ref_gemm    = GemmInstance{};
    auto ref_invoker = ref_gemm.MakeInvoker();

    auto ref_argument = ref_gemm.MakeArgument(A, B, C, a_element_op, b_element_op, c_element_op);

    ref_invoker.Run(ref_argument);
}

template <typename DeviceGemmPtr_,
          typename ADataType,
          typename BDataType,
          typename CDataType,
          typename AElementwiseOperation,
          typename BElementwiseOperation,
          typename CElementwiseOperation>
Jianfeng Yan's avatar
Jianfeng Yan committed
58
bool RunDeviceGEMM(DeviceGemmPtr_& gemmPtr,
Anthony Chang's avatar
Anthony Chang committed
59
60
61
62
63
64
                   const ck::gemm_util::GemmParams& params,
                   const Tensor<ADataType>& A,
                   const Tensor<BDataType>& B,
                   Tensor<CDataType>& C,
                   AElementwiseOperation a_element_op,
                   BElementwiseOperation b_element_op,
65
66
                   CElementwiseOperation c_element_op,
                   bool time_kernel)
Anthony Chang's avatar
Anthony Chang committed
67
{
68
69
70
    DeviceMem a_m_k_device_buf(sizeof(ADataType) * A.mDesc.GetElementSpaceSize());
    DeviceMem b_k_n_device_buf(sizeof(BDataType) * B.mDesc.GetElementSpaceSize());
    DeviceMem c_m_n_device_buf(sizeof(CDataType) * C.mDesc.GetElementSpaceSize());
Anthony Chang's avatar
Anthony Chang committed
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86

    auto invoker_ptr = gemmPtr->MakeInvokerPointer();
    auto argument_ptr =
        gemmPtr->MakeArgumentPointer(static_cast<ADataType*>(a_m_k_device_buf.GetDeviceBuffer()),
                                     static_cast<BDataType*>(b_k_n_device_buf.GetDeviceBuffer()),
                                     static_cast<CDataType*>(c_m_n_device_buf.GetDeviceBuffer()),
                                     params.M,
                                     params.N,
                                     params.K,
                                     params.StrideA,
                                     params.StrideB,
                                     params.StrideC,
                                     a_element_op,
                                     b_element_op,
                                     c_element_op);

Jianfeng Yan's avatar
Jianfeng Yan committed
87
    if(gemmPtr->IsSupportedArgument(argument_ptr.get()))
Anthony Chang's avatar
Anthony Chang committed
88
    {
Jianfeng Yan's avatar
Jianfeng Yan committed
89
90
        a_m_k_device_buf.ToDevice(A.mData.data());
        b_k_n_device_buf.ToDevice(B.mData.data());
91
92
93
94
95
96
97
98
99
100
101
102
103
104
        float ave_time = invoker_ptr->Run(argument_ptr.get(), StreamConfig{nullptr, time_kernel});

        std::size_t flop      = std::size_t(2) * params.M * params.N * params.K;
        std::size_t num_btype = sizeof(ADataType) * params.M * params.K +
                                sizeof(BDataType) * params.K * params.N +
                                sizeof(CDataType) * params.M * params.N;

        float tflops = static_cast<float>(flop) / 1.E9 / ave_time;

        float gb_per_sec = num_btype / 1.E6 / ave_time;

        std::cout << "Perf: " << ave_time << " ms, " << tflops << " TFlops, " << gb_per_sec
                  << " GB/s, " << std::endl;

Jianfeng Yan's avatar
Jianfeng Yan committed
105
106
107
        c_m_n_device_buf.FromDevice(C.mData.data());

        return true;
Anthony Chang's avatar
Anthony Chang committed
108
    }
Jianfeng Yan's avatar
Jianfeng Yan committed
109
110
111
112
113
    else
    {
        std::cout << "device_gemm with the specified compilation parameters does "
                     "not support this GEMM problem"
                  << std::endl;
Anthony Chang's avatar
Anthony Chang committed
114

Jianfeng Yan's avatar
Jianfeng Yan committed
115
116
        return false;
    }
Anthony Chang's avatar
Anthony Chang committed
117
118
}

119
template <typename AccDataType>
Anthony Chang's avatar
Anthony Chang committed
120
121
struct TestGemm
{
122
123
124
125
126
127
    template <typename ADataType,
              typename BDataType,
              typename CDataType,
              typename ALayout,
              typename BLayout,
              typename CLayout>
Anthony Chang's avatar
Anthony Chang committed
128
129
130
131
    auto PrepareGemmTensor(const ck::gemm_util::GemmParams& params)
    {
        auto f_host_tensor_descriptor =
            [](std::size_t row, std::size_t col, std::size_t stride, auto layout) {
132
133
                using namespace ck::literals;

Anthony Chang's avatar
Anthony Chang committed
134
135
                if(std::is_same<decltype(layout), ck::tensor_layout::gemm::RowMajor>::value)
                {
136
                    return HostTensorDescriptor({row, col}, {stride, 1_uz});
Anthony Chang's avatar
Anthony Chang committed
137
138
139
                }
                else
                {
140
                    return HostTensorDescriptor({row, col}, {1_uz, stride});
Anthony Chang's avatar
Anthony Chang committed
141
142
143
144
145
146
147
148
149
150
151
152
                }
            };

        Tensor<ADataType> a_m_k(
            f_host_tensor_descriptor(params.M, params.K, params.StrideA, ALayout{}));
        Tensor<BDataType> b_k_n(
            f_host_tensor_descriptor(params.K, params.N, params.StrideB, BLayout{}));
        Tensor<CDataType> c_m_n_host_result(
            f_host_tensor_descriptor(params.M, params.N, params.StrideC, CLayout{}));
        Tensor<CDataType> c_m_n_device_result(
            f_host_tensor_descriptor(params.M, params.N, params.StrideC, CLayout{}));

153
        auto f_generate_tensor_value = [](auto& tensor, auto type) {
Anthony Chang's avatar
Anthony Chang committed
154
155
            using dataType = decltype(type);

156
            tensor.GenerateTensorValue(GeneratorTensor_2<dataType>{-5, 5});
Anthony Chang's avatar
Anthony Chang committed
157
158
159
160
161
        };

        f_generate_tensor_value(a_m_k, ADataType{});
        f_generate_tensor_value(b_k_n, BDataType{});

162
163
164
165
        std::cout << "a_m_k: " << a_m_k.mDesc << std::endl;
        std::cout << "b_k_n: " << b_k_n.mDesc << std::endl;
        std::cout << "c_m_n: " << c_m_n_host_result.mDesc << std::endl;

Anthony Chang's avatar
Anthony Chang committed
166
167
168
        return std::make_tuple(a_m_k, b_k_n, c_m_n_host_result, c_m_n_device_result);
    }

169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
    template <template <class...> class DeviceGemmPtr_,
              typename ALayout,
              typename BLayout,
              typename CLayout,
              typename ADataType,
              typename BDataType,
              typename CDataType,
              typename AElementwiseOperation,
              typename BElementwiseOperation,
              typename CElementwiseOperation>
    auto operator()(DeviceGemmPtr_<ALayout,
                                   BLayout,
                                   CLayout,
                                   ADataType,
                                   BDataType,
                                   CDataType,
                                   AElementwiseOperation,
                                   BElementwiseOperation,
                                   CElementwiseOperation>* gemmPtr,
                    const GemmParams& params = GemmParams{},
                    bool do_verification     = true,
                    bool time_kernel         = false)
Anthony Chang's avatar
Anthony Chang committed
191
192
193
194
195
    {
        std::cout << "ALayout = " << ALayout{}.name << ", BLayout = " << BLayout{}.name
                  << ", CLayout = " << CLayout{}.name << std::endl;
        std::cout << gemmPtr->GetTypeString() << std::endl;

196
197
        auto host_tensors =
            PrepareGemmTensor<ADataType, BDataType, CDataType, ALayout, BLayout, CLayout>(params);
Anthony Chang's avatar
Anthony Chang committed
198
199
200
201
202
203
204
205
206
207
208
209
210
211

        const Tensor<ADataType>& a  = std::get<0>(host_tensors);
        const Tensor<BDataType>& b  = std::get<1>(host_tensors);
        Tensor<CDataType>& c_host   = std::get<2>(host_tensors);
        Tensor<CDataType>& c_device = std::get<3>(host_tensors);

        auto a_element_op = AElementwiseOperation{};
        auto b_element_op = BElementwiseOperation{};
        auto c_element_op = CElementwiseOperation{};

        using ReferenceGemmInstance =
            ck::tensor_operation::host::ReferenceGemm<ADataType,
                                                      BDataType,
                                                      CDataType,
212
                                                      AccDataType,
Anthony Chang's avatar
Anthony Chang committed
213
214
215
                                                      AElementwiseOperation,
                                                      BElementwiseOperation,
                                                      CElementwiseOperation>;
216
217
218
219
220
221

        if(do_verification)
        {
            ck::gemm_util::RunHostGEMM<ReferenceGemmInstance>(
                a, b, c_host, a_element_op, b_element_op, c_element_op);
        }
Anthony Chang's avatar
Anthony Chang committed
222
223

        // Act
Jianfeng Yan's avatar
Jianfeng Yan committed
224
        bool is_supported = ck::gemm_util::RunDeviceGEMM(
225
            gemmPtr, params, a, b, c_device, a_element_op, b_element_op, c_element_op, time_kernel);
Anthony Chang's avatar
Anthony Chang committed
226

227
        if(is_supported && do_verification)
Anthony Chang's avatar
Anthony Chang committed
228
        {
Jianfeng Yan's avatar
Jianfeng Yan committed
229
230
231
232
            // Assert
            bool res = false;
            if(std::is_same<CDataType, float>::value)
            {
233
                res = ck::utils::check_err(c_device, c_host);
Jianfeng Yan's avatar
Jianfeng Yan committed
234
235
236
237
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
            else if(std::is_same<CDataType, ck::half_t>::value)
            {
238
                res = ck::utils::check_err(c_device, c_host);
Jianfeng Yan's avatar
Jianfeng Yan committed
239
240
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
Chao Liu's avatar
Chao Liu committed
241
242
            else if(std::is_same<CDataType, ck::bhalf_t>::value)
            {
243
                res = ck::utils::check_err(c_device, c_host);
Chao Liu's avatar
Chao Liu committed
244
245
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
Jianfeng Yan's avatar
Jianfeng Yan committed
246
247
            else if(std::is_same<CDataType, int8_t>::value)
            {
248
                res = ck::utils::check_err(c_device, c_host);
Jianfeng Yan's avatar
Jianfeng Yan committed
249
250
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
251
252
            else if(std::is_same<CDataType, double>::value)
            {
253
                res = ck::utils::check_err(c_device, c_host);
254
255
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
Jianfeng Yan's avatar
Jianfeng Yan committed
256
257

            return res;
Anthony Chang's avatar
Anthony Chang committed
258
        }
Jianfeng Yan's avatar
Jianfeng Yan committed
259
        else
Anthony Chang's avatar
Anthony Chang committed
260
        {
Jianfeng Yan's avatar
Jianfeng Yan committed
261
            return true;
Anthony Chang's avatar
Anthony Chang committed
262
263
264
265
266
267
        }
    }
};

} // namespace gemm_util
} // namespace ck