gemm_util.hpp 10.3 KB
Newer Older
Chao Liu's avatar
Chao Liu committed
1
2
3
// SPDX-License-Identifier: MIT
// Copyright (c) 2018-2022, Advanced Micro Devices, Inc. All rights reserved.

Chao Liu's avatar
Chao Liu committed
4
#pragma once
Anthony Chang's avatar
Anthony Chang committed
5

Chao Liu's avatar
Chao Liu committed
6
7
8
#include "ck/ck.hpp"
#include "ck/tensor_operation/gpu/device/tensor_layout.hpp"
#include "ck/library/utility/check_err.hpp"
9
10
11
#include "ck/library/utility/device_memory.hpp"
#include "ck/library/utility/host_tensor.hpp"
#include "ck/library/utility/host_tensor_generator.hpp"
Chao Liu's avatar
Chao Liu committed
12
#include "ck/library/reference_tensor_operation/cpu/reference_gemm.hpp"
Anthony Chang's avatar
Anthony Chang committed
13
14
15
16
17
18

namespace ck {
namespace gemm_util {

struct GemmParams
{
19
20
21
    ck::index_t M = 1024;
    ck::index_t N = 1024;
    ck::index_t K = 1024;
Anthony Chang's avatar
Anthony Chang committed
22

23
24
25
    ck::index_t StrideA = 1024;
    ck::index_t StrideB = 1024;
    ck::index_t StrideC = 1024;
Anthony Chang's avatar
Anthony Chang committed
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
};

template <typename GemmInstance,
          typename ADataType,
          typename BDataType,
          typename CDataType,
          typename AElementwiseOperation,
          typename BElementwiseOperation,
          typename CElementwiseOperation>
void RunHostGEMM(const Tensor<ADataType>& A,
                 const Tensor<BDataType>& B,
                 Tensor<CDataType>& C,
                 AElementwiseOperation a_element_op,
                 BElementwiseOperation b_element_op,
                 CElementwiseOperation c_element_op)
{
    auto ref_gemm    = GemmInstance{};
    auto ref_invoker = ref_gemm.MakeInvoker();

    auto ref_argument = ref_gemm.MakeArgument(A, B, C, a_element_op, b_element_op, c_element_op);

    ref_invoker.Run(ref_argument);
}

template <typename DeviceGemmPtr_,
          typename ADataType,
          typename BDataType,
          typename CDataType,
          typename AElementwiseOperation,
          typename BElementwiseOperation,
          typename CElementwiseOperation>
Jianfeng Yan's avatar
Jianfeng Yan committed
57
bool RunDeviceGEMM(DeviceGemmPtr_& gemmPtr,
Anthony Chang's avatar
Anthony Chang committed
58
59
60
61
62
63
                   const ck::gemm_util::GemmParams& params,
                   const Tensor<ADataType>& A,
                   const Tensor<BDataType>& B,
                   Tensor<CDataType>& C,
                   AElementwiseOperation a_element_op,
                   BElementwiseOperation b_element_op,
64
65
                   CElementwiseOperation c_element_op,
                   bool time_kernel)
Anthony Chang's avatar
Anthony Chang committed
66
{
67
68
69
    DeviceMem a_m_k_device_buf(sizeof(ADataType) * A.mDesc.GetElementSpaceSize());
    DeviceMem b_k_n_device_buf(sizeof(BDataType) * B.mDesc.GetElementSpaceSize());
    DeviceMem c_m_n_device_buf(sizeof(CDataType) * C.mDesc.GetElementSpaceSize());
Anthony Chang's avatar
Anthony Chang committed
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85

    auto invoker_ptr = gemmPtr->MakeInvokerPointer();
    auto argument_ptr =
        gemmPtr->MakeArgumentPointer(static_cast<ADataType*>(a_m_k_device_buf.GetDeviceBuffer()),
                                     static_cast<BDataType*>(b_k_n_device_buf.GetDeviceBuffer()),
                                     static_cast<CDataType*>(c_m_n_device_buf.GetDeviceBuffer()),
                                     params.M,
                                     params.N,
                                     params.K,
                                     params.StrideA,
                                     params.StrideB,
                                     params.StrideC,
                                     a_element_op,
                                     b_element_op,
                                     c_element_op);

Jianfeng Yan's avatar
Jianfeng Yan committed
86
    if(gemmPtr->IsSupportedArgument(argument_ptr.get()))
Anthony Chang's avatar
Anthony Chang committed
87
    {
Jianfeng Yan's avatar
Jianfeng Yan committed
88
89
        a_m_k_device_buf.ToDevice(A.mData.data());
        b_k_n_device_buf.ToDevice(B.mData.data());
90
91
92
93
94
95
96
97
98
99
100
101
102
103
        float ave_time = invoker_ptr->Run(argument_ptr.get(), StreamConfig{nullptr, time_kernel});

        std::size_t flop      = std::size_t(2) * params.M * params.N * params.K;
        std::size_t num_btype = sizeof(ADataType) * params.M * params.K +
                                sizeof(BDataType) * params.K * params.N +
                                sizeof(CDataType) * params.M * params.N;

        float tflops = static_cast<float>(flop) / 1.E9 / ave_time;

        float gb_per_sec = num_btype / 1.E6 / ave_time;

        std::cout << "Perf: " << ave_time << " ms, " << tflops << " TFlops, " << gb_per_sec
                  << " GB/s, " << std::endl;

Jianfeng Yan's avatar
Jianfeng Yan committed
104
105
106
        c_m_n_device_buf.FromDevice(C.mData.data());

        return true;
Anthony Chang's avatar
Anthony Chang committed
107
    }
Jianfeng Yan's avatar
Jianfeng Yan committed
108
109
110
111
112
    else
    {
        std::cout << "device_gemm with the specified compilation parameters does "
                     "not support this GEMM problem"
                  << std::endl;
Anthony Chang's avatar
Anthony Chang committed
113

Jianfeng Yan's avatar
Jianfeng Yan committed
114
115
        return false;
    }
Anthony Chang's avatar
Anthony Chang committed
116
117
}

118
template <typename AccDataType>
Anthony Chang's avatar
Anthony Chang committed
119
120
struct TestGemm
{
121
122
123
124
125
126
    template <typename ADataType,
              typename BDataType,
              typename CDataType,
              typename ALayout,
              typename BLayout,
              typename CLayout>
Anthony Chang's avatar
Anthony Chang committed
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
    auto PrepareGemmTensor(const ck::gemm_util::GemmParams& params)
    {
        auto f_host_tensor_descriptor =
            [](std::size_t row, std::size_t col, std::size_t stride, auto layout) {
                if(std::is_same<decltype(layout), ck::tensor_layout::gemm::RowMajor>::value)
                {
                    return HostTensorDescriptor(std::vector<std::size_t>({row, col}),
                                                std::vector<std::size_t>({stride, 1}));
                }
                else
                {
                    return HostTensorDescriptor(std::vector<std::size_t>({row, col}),
                                                std::vector<std::size_t>({1, stride}));
                }
            };

        Tensor<ADataType> a_m_k(
            f_host_tensor_descriptor(params.M, params.K, params.StrideA, ALayout{}));
        Tensor<BDataType> b_k_n(
            f_host_tensor_descriptor(params.K, params.N, params.StrideB, BLayout{}));
        Tensor<CDataType> c_m_n_host_result(
            f_host_tensor_descriptor(params.M, params.N, params.StrideC, CLayout{}));
        Tensor<CDataType> c_m_n_device_result(
            f_host_tensor_descriptor(params.M, params.N, params.StrideC, CLayout{}));

152
        auto f_generate_tensor_value = [](auto& tensor, auto type) {
Anthony Chang's avatar
Anthony Chang committed
153
154
            using dataType = decltype(type);

155
            tensor.GenerateTensorValue(GeneratorTensor_2<dataType>{-5, 5});
Anthony Chang's avatar
Anthony Chang committed
156
157
158
159
160
        };

        f_generate_tensor_value(a_m_k, ADataType{});
        f_generate_tensor_value(b_k_n, BDataType{});

161
162
163
164
        std::cout << "a_m_k: " << a_m_k.mDesc << std::endl;
        std::cout << "b_k_n: " << b_k_n.mDesc << std::endl;
        std::cout << "c_m_n: " << c_m_n_host_result.mDesc << std::endl;

Anthony Chang's avatar
Anthony Chang committed
165
166
167
        return std::make_tuple(a_m_k, b_k_n, c_m_n_host_result, c_m_n_device_result);
    }

168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
    template <template <class...> class DeviceGemmPtr_,
              typename ALayout,
              typename BLayout,
              typename CLayout,
              typename ADataType,
              typename BDataType,
              typename CDataType,
              typename AElementwiseOperation,
              typename BElementwiseOperation,
              typename CElementwiseOperation>
    auto operator()(DeviceGemmPtr_<ALayout,
                                   BLayout,
                                   CLayout,
                                   ADataType,
                                   BDataType,
                                   CDataType,
                                   AElementwiseOperation,
                                   BElementwiseOperation,
                                   CElementwiseOperation>* gemmPtr,
                    const GemmParams& params = GemmParams{},
                    bool do_verification     = true,
                    bool time_kernel         = false)
Anthony Chang's avatar
Anthony Chang committed
190
191
192
193
194
    {
        std::cout << "ALayout = " << ALayout{}.name << ", BLayout = " << BLayout{}.name
                  << ", CLayout = " << CLayout{}.name << std::endl;
        std::cout << gemmPtr->GetTypeString() << std::endl;

195
196
        auto host_tensors =
            PrepareGemmTensor<ADataType, BDataType, CDataType, ALayout, BLayout, CLayout>(params);
Anthony Chang's avatar
Anthony Chang committed
197
198
199
200
201
202
203
204
205
206
207
208
209
210

        const Tensor<ADataType>& a  = std::get<0>(host_tensors);
        const Tensor<BDataType>& b  = std::get<1>(host_tensors);
        Tensor<CDataType>& c_host   = std::get<2>(host_tensors);
        Tensor<CDataType>& c_device = std::get<3>(host_tensors);

        auto a_element_op = AElementwiseOperation{};
        auto b_element_op = BElementwiseOperation{};
        auto c_element_op = CElementwiseOperation{};

        using ReferenceGemmInstance =
            ck::tensor_operation::host::ReferenceGemm<ADataType,
                                                      BDataType,
                                                      CDataType,
211
                                                      AccDataType,
Anthony Chang's avatar
Anthony Chang committed
212
213
214
                                                      AElementwiseOperation,
                                                      BElementwiseOperation,
                                                      CElementwiseOperation>;
215
216
217
218
219
220

        if(do_verification)
        {
            ck::gemm_util::RunHostGEMM<ReferenceGemmInstance>(
                a, b, c_host, a_element_op, b_element_op, c_element_op);
        }
Anthony Chang's avatar
Anthony Chang committed
221
222

        // Act
Jianfeng Yan's avatar
Jianfeng Yan committed
223
        bool is_supported = ck::gemm_util::RunDeviceGEMM(
224
            gemmPtr, params, a, b, c_device, a_element_op, b_element_op, c_element_op, time_kernel);
Anthony Chang's avatar
Anthony Chang committed
225

226
        if(is_supported && do_verification)
Anthony Chang's avatar
Anthony Chang committed
227
        {
Jianfeng Yan's avatar
Jianfeng Yan committed
228
229
230
231
232
233
234
235
236
237
238
239
            // Assert
            bool res = false;
            if(std::is_same<CDataType, float>::value)
            {
                res = ck::utils::check_err(c_device.mData, c_host.mData);
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
            else if(std::is_same<CDataType, ck::half_t>::value)
            {
                res = ck::utils::check_err(c_device.mData, c_host.mData);
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
Chao Liu's avatar
Chao Liu committed
240
241
242
243
244
            else if(std::is_same<CDataType, ck::bhalf_t>::value)
            {
                res = ck::utils::check_err(c_device.mData, c_host.mData);
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
Jianfeng Yan's avatar
Jianfeng Yan committed
245
246
247
248
249
            else if(std::is_same<CDataType, int8_t>::value)
            {
                res = ck::utils::check_err(c_device.mData, c_host.mData);
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
250
251
252
253
254
            else if(std::is_same<CDataType, double>::value)
            {
                res = ck::utils::check_err(c_device.mData, c_host.mData);
                std::cout << (res ? "SUCCESS" : "FAILURE") << std::endl;
            }
Jianfeng Yan's avatar
Jianfeng Yan committed
255
256

            return res;
Anthony Chang's avatar
Anthony Chang committed
257
        }
Jianfeng Yan's avatar
Jianfeng Yan committed
258
        else
Anthony Chang's avatar
Anthony Chang committed
259
        {
Jianfeng Yan's avatar
Jianfeng Yan committed
260
            return true;
Anthony Chang's avatar
Anthony Chang committed
261
262
263
264
265
266
        }
    }
};

} // namespace gemm_util
} // namespace ck