gemm_util.hpp 10.3 KB
Newer Older
Chao Liu's avatar
Chao Liu committed
1
2
3
// SPDX-License-Identifier: MIT
// Copyright (c) 2018-2022, Advanced Micro Devices, Inc. All rights reserved.

Chao Liu's avatar
Chao Liu committed
4
#pragma once
Anthony Chang's avatar
Anthony Chang committed
5

Chao Liu's avatar
Chao Liu committed
6
7
8
#include "ck/ck.hpp"
#include "ck/tensor_operation/gpu/device/tensor_layout.hpp"
#include "ck/library/utility/check_err.hpp"
9
10
11
#include "ck/library/utility/device_memory.hpp"
#include "ck/library/utility/host_tensor.hpp"
#include "ck/library/utility/host_tensor_generator.hpp"
12
#include "ck/library/utility/literals.hpp"
Chao Liu's avatar
Chao Liu committed
13
#include "ck/library/reference_tensor_operation/cpu/reference_gemm.hpp"
Anthony Chang's avatar
Anthony Chang committed
14
15
16
17
18
19

namespace ck {
namespace gemm_util {

struct GemmParams
{
20
21
22
    ck::index_t M = 1024;
    ck::index_t N = 1024;
    ck::index_t K = 1024;
Anthony Chang's avatar
Anthony Chang committed
23

24
25
26
    ck::index_t StrideA = 1024;
    ck::index_t StrideB = 1024;
    ck::index_t StrideC = 1024;
Anthony Chang's avatar
Anthony Chang committed
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
};

template <typename GemmInstance,
          typename ADataType,
          typename BDataType,
          typename CDataType,
          typename AElementwiseOperation,
          typename BElementwiseOperation,
          typename CElementwiseOperation>
void RunHostGEMM(const Tensor<ADataType>& A,
                 const Tensor<BDataType>& B,
                 Tensor<CDataType>& C,
                 AElementwiseOperation a_element_op,
                 BElementwiseOperation b_element_op,
                 CElementwiseOperation c_element_op)
{
    auto ref_gemm    = GemmInstance{};
    auto ref_invoker = ref_gemm.MakeInvoker();

    auto ref_argument = ref_gemm.MakeArgument(A, B, C, a_element_op, b_element_op, c_element_op);

    ref_invoker.Run(ref_argument);
}

template <typename DeviceGemmPtr_,
          typename ADataType,
          typename BDataType,
          typename CDataType,
          typename AElementwiseOperation,
          typename BElementwiseOperation,
          typename CElementwiseOperation>
Jianfeng Yan's avatar
Jianfeng Yan committed
58
bool RunDeviceGEMM(DeviceGemmPtr_& gemmPtr,
Anthony Chang's avatar
Anthony Chang committed
59
60
61
62
63
64
                   const ck::gemm_util::GemmParams& params,
                   const Tensor<ADataType>& A,
                   const Tensor<BDataType>& B,
                   Tensor<CDataType>& C,
                   AElementwiseOperation a_element_op,
                   BElementwiseOperation b_element_op,
65
                   CElementwiseOperation c_element_op,
Adam Osewski's avatar
Adam Osewski committed
66
67
                   bool time_kernel,
                   ck::index_t b2c_M01)
Anthony Chang's avatar
Anthony Chang committed
68
{
69
70
71
    DeviceMem a_m_k_device_buf(sizeof(ADataType) * A.mDesc.GetElementSpaceSize());
    DeviceMem b_k_n_device_buf(sizeof(BDataType) * B.mDesc.GetElementSpaceSize());
    DeviceMem c_m_n_device_buf(sizeof(CDataType) * C.mDesc.GetElementSpaceSize());
Anthony Chang's avatar
Anthony Chang committed
72
73
74
75
76
77
78
79
80
81
82
83
84
85

    auto invoker_ptr = gemmPtr->MakeInvokerPointer();
    auto argument_ptr =
        gemmPtr->MakeArgumentPointer(static_cast<ADataType*>(a_m_k_device_buf.GetDeviceBuffer()),
                                     static_cast<BDataType*>(b_k_n_device_buf.GetDeviceBuffer()),
                                     static_cast<CDataType*>(c_m_n_device_buf.GetDeviceBuffer()),
                                     params.M,
                                     params.N,
                                     params.K,
                                     params.StrideA,
                                     params.StrideB,
                                     params.StrideC,
                                     a_element_op,
                                     b_element_op,
Adam Osewski's avatar
Adam Osewski committed
86
87
                                     c_element_op,
                                     b2c_M01);
Anthony Chang's avatar
Anthony Chang committed
88

Jianfeng Yan's avatar
Jianfeng Yan committed
89
    if(gemmPtr->IsSupportedArgument(argument_ptr.get()))
Anthony Chang's avatar
Anthony Chang committed
90
    {
Jianfeng Yan's avatar
Jianfeng Yan committed
91
92
        a_m_k_device_buf.ToDevice(A.mData.data());
        b_k_n_device_buf.ToDevice(B.mData.data());
93
94
95
96
97
98
99
100
101
102
103
104
105
106
        float ave_time = invoker_ptr->Run(argument_ptr.get(), StreamConfig{nullptr, time_kernel});

        std::size_t flop      = std::size_t(2) * params.M * params.N * params.K;
        std::size_t num_btype = sizeof(ADataType) * params.M * params.K +
                                sizeof(BDataType) * params.K * params.N +
                                sizeof(CDataType) * params.M * params.N;

        float tflops = static_cast<float>(flop) / 1.E9 / ave_time;

        float gb_per_sec = num_btype / 1.E6 / ave_time;

        std::cout << "Perf: " << ave_time << " ms, " << tflops << " TFlops, " << gb_per_sec
                  << " GB/s, " << std::endl;

Jianfeng Yan's avatar
Jianfeng Yan committed
107
108
109
        c_m_n_device_buf.FromDevice(C.mData.data());

        return true;
Anthony Chang's avatar
Anthony Chang committed
110
    }
Jianfeng Yan's avatar
Jianfeng Yan committed
111
112
113
114
115
    else
    {
        std::cout << "device_gemm with the specified compilation parameters does "
                     "not support this GEMM problem"
                  << std::endl;
Anthony Chang's avatar
Anthony Chang committed
116

Jianfeng Yan's avatar
Jianfeng Yan committed
117
118
        return false;
    }
Anthony Chang's avatar
Anthony Chang committed
119
120
}

121
template <typename AccDataType>
Anthony Chang's avatar
Anthony Chang committed
122
123
struct TestGemm
{
124
125
126
127
128
129
    template <typename ADataType,
              typename BDataType,
              typename CDataType,
              typename ALayout,
              typename BLayout,
              typename CLayout>
Anthony Chang's avatar
Anthony Chang committed
130
131
132
133
    auto PrepareGemmTensor(const ck::gemm_util::GemmParams& params)
    {
        auto f_host_tensor_descriptor =
            [](std::size_t row, std::size_t col, std::size_t stride, auto layout) {
134
135
                using namespace ck::literals;

Anthony Chang's avatar
Anthony Chang committed
136
137
                if(std::is_same<decltype(layout), ck::tensor_layout::gemm::RowMajor>::value)
                {
138
                    return HostTensorDescriptor({row, col}, {stride, 1_uz});
Anthony Chang's avatar
Anthony Chang committed
139
140
141
                }
                else
                {
142
                    return HostTensorDescriptor({row, col}, {1_uz, stride});
Anthony Chang's avatar
Anthony Chang committed
143
144
145
146
147
148
149
150
151
152
153
154
                }
            };

        Tensor<ADataType> a_m_k(
            f_host_tensor_descriptor(params.M, params.K, params.StrideA, ALayout{}));
        Tensor<BDataType> b_k_n(
            f_host_tensor_descriptor(params.K, params.N, params.StrideB, BLayout{}));
        Tensor<CDataType> c_m_n_host_result(
            f_host_tensor_descriptor(params.M, params.N, params.StrideC, CLayout{}));
        Tensor<CDataType> c_m_n_device_result(
            f_host_tensor_descriptor(params.M, params.N, params.StrideC, CLayout{}));

155
        auto f_generate_tensor_value = [](auto& tensor, auto type) {
Anthony Chang's avatar
Anthony Chang committed
156
157
            using dataType = decltype(type);

158
            tensor.GenerateTensorValue(GeneratorTensor_2<dataType>{-5, 5});
Anthony Chang's avatar
Anthony Chang committed
159
160
161
162
163
        };

        f_generate_tensor_value(a_m_k, ADataType{});
        f_generate_tensor_value(b_k_n, BDataType{});

164
165
166
167
        std::cout << "a_m_k: " << a_m_k.mDesc << std::endl;
        std::cout << "b_k_n: " << b_k_n.mDesc << std::endl;
        std::cout << "c_m_n: " << c_m_n_host_result.mDesc << std::endl;

Anthony Chang's avatar
Anthony Chang committed
168
169
170
        return std::make_tuple(a_m_k, b_k_n, c_m_n_host_result, c_m_n_device_result);
    }

171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
    template <template <class...> class DeviceGemmPtr_,
              typename ALayout,
              typename BLayout,
              typename CLayout,
              typename ADataType,
              typename BDataType,
              typename CDataType,
              typename AElementwiseOperation,
              typename BElementwiseOperation,
              typename CElementwiseOperation>
    auto operator()(DeviceGemmPtr_<ALayout,
                                   BLayout,
                                   CLayout,
                                   ADataType,
                                   BDataType,
                                   CDataType,
                                   AElementwiseOperation,
                                   BElementwiseOperation,
                                   CElementwiseOperation>* gemmPtr,
                    const GemmParams& params = GemmParams{},
                    bool do_verification     = true,
Adam Osewski's avatar
Adam Osewski committed
192
193
                    bool time_kernel         = false,
                    ck::index_t b2c_M01      = 8)
Anthony Chang's avatar
Anthony Chang committed
194
195
196
197
198
    {
        std::cout << "ALayout = " << ALayout{}.name << ", BLayout = " << BLayout{}.name
                  << ", CLayout = " << CLayout{}.name << std::endl;
        std::cout << gemmPtr->GetTypeString() << std::endl;

199
200
        auto host_tensors =
            PrepareGemmTensor<ADataType, BDataType, CDataType, ALayout, BLayout, CLayout>(params);
Anthony Chang's avatar
Anthony Chang committed
201
202
203
204
205
206
207
208
209
210
211
212
213
214

        const Tensor<ADataType>& a  = std::get<0>(host_tensors);
        const Tensor<BDataType>& b  = std::get<1>(host_tensors);
        Tensor<CDataType>& c_host   = std::get<2>(host_tensors);
        Tensor<CDataType>& c_device = std::get<3>(host_tensors);

        auto a_element_op = AElementwiseOperation{};
        auto b_element_op = BElementwiseOperation{};
        auto c_element_op = CElementwiseOperation{};

        using ReferenceGemmInstance =
            ck::tensor_operation::host::ReferenceGemm<ADataType,
                                                      BDataType,
                                                      CDataType,
215
                                                      AccDataType,
Anthony Chang's avatar
Anthony Chang committed
216
217
218
                                                      AElementwiseOperation,
                                                      BElementwiseOperation,
                                                      CElementwiseOperation>;
219
220
221
222
223
224

        if(do_verification)
        {
            ck::gemm_util::RunHostGEMM<ReferenceGemmInstance>(
                a, b, c_host, a_element_op, b_element_op, c_element_op);
        }
Anthony Chang's avatar
Anthony Chang committed
225
226

        // Act
Jianfeng Yan's avatar
Jianfeng Yan committed
227
        bool is_supported = ck::gemm_util::RunDeviceGEMM(
Adam Osewski's avatar
Adam Osewski committed
228
229
            gemmPtr, params, a, b, c_device, a_element_op, b_element_op, c_element_op, time_kernel,
            b2c_M01);
Anthony Chang's avatar
Anthony Chang committed
230

231
        if(is_supported && do_verification)
Anthony Chang's avatar
Anthony Chang committed
232
        {
Jianfeng Yan's avatar
Jianfeng Yan committed
233
234
235
236
            // Assert
            bool res = false;
            if(std::is_same<CDataType, float>::value)
            {
237
                res = ck::utils::check_err(c_device, c_host);
Jianfeng Yan's avatar
Jianfeng Yan committed
238
239
240
241
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
            else if(std::is_same<CDataType, ck::half_t>::value)
            {
242
                res = ck::utils::check_err(c_device, c_host);
Jianfeng Yan's avatar
Jianfeng Yan committed
243
244
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
Chao Liu's avatar
Chao Liu committed
245
246
            else if(std::is_same<CDataType, ck::bhalf_t>::value)
            {
247
                res = ck::utils::check_err(c_device, c_host);
Chao Liu's avatar
Chao Liu committed
248
249
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
Jianfeng Yan's avatar
Jianfeng Yan committed
250
251
            else if(std::is_same<CDataType, int8_t>::value)
            {
252
                res = ck::utils::check_err(c_device, c_host);
Jianfeng Yan's avatar
Jianfeng Yan committed
253
254
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
255
256
            else if(std::is_same<CDataType, double>::value)
            {
257
                res = ck::utils::check_err(c_device, c_host);
258
259
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
Jianfeng Yan's avatar
Jianfeng Yan committed
260
261

            return res;
Anthony Chang's avatar
Anthony Chang committed
262
        }
Jianfeng Yan's avatar
Jianfeng Yan committed
263
        else
Anthony Chang's avatar
Anthony Chang committed
264
        {
Jianfeng Yan's avatar
Jianfeng Yan committed
265
            return true;
Anthony Chang's avatar
Anthony Chang committed
266
267
268
269
270
271
        }
    }
};

} // namespace gemm_util
} // namespace ck