utils.cuh 22.9 KB
Newer Older
lijian6's avatar
lijian6 committed
1
#include "hip/hip_runtime.h"
Chenggang Zhao's avatar
Chenggang Zhao committed
2
#pragma once
lijian6's avatar
lijian6 committed
3
#include "configs.cuh"
Chenggang Zhao's avatar
Chenggang Zhao committed
4
5
#include "exception.cuh"

lijian6's avatar
lijian6 committed
6
7
8
9
10
11
12
13
14
15
16
17
18
#define UNROLLED_WARP_COPY(UNROLL_FACTOR, LANE_ID, N, DST, SRC, LD_FUNC, ST_FUNC)                  \
    {                                                                                              \
        constexpr int kLoopStride = kWarpSize * (UNROLL_FACTOR);                                   \
        typename std::remove_reference<decltype(LD_FUNC((SRC) + 0))>::type                         \
             unrolled_values[(UNROLL_FACTOR)];                                                     \
        auto __src = (SRC);                                                                        \
        auto __dst = (DST);                                                                        \
        for (int __i = (LANE_ID); __i < ((N) / kLoopStride) * kLoopStride; __i += kLoopStride) {   \
            _Pragma("unroll") for (int __j = 0; __j < (UNROLL_FACTOR); ++__j)                      \
                unrolled_values[__j] = LD_FUNC(__src + __i + __j * kWarpSize);                     \
            _Pragma("unroll") for (int __j = 0; __j < (UNROLL_FACTOR); ++__j)                      \
                ST_FUNC(__dst + __i + __j * kWarpSize, unrolled_values[__j]);                      \
        }                                                                                          \
lijian6's avatar
lijian6 committed
19
20
        for (int __i = ((N) / kLoopStride) * kLoopStride + (LANE_ID); __i < (N); __i += kWarpSize) \
            ST_FUNC(__dst + __i, LD_FUNC(__src + __i)); \
lijian6's avatar
lijian6 committed
21
22
    }

lishen's avatar
lishen committed
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
#define UNROLLED_WARP_COPY_LL(UNROLL_FACTOR, LANE_ID, N, DST, SRC, LD_FUNC, ST_FUNC)                                                        \
    {                                                                                                                                       \
        constexpr int kLoopStride = kWarpSize * (UNROLL_FACTOR);                                                                            \
        typename std::remove_reference<decltype(LD_FUNC((SRC) + 0))>::type unrolled_values[(UNROLL_FACTOR)];                                \
        auto __src = (SRC);                                                                                                                 \
        auto __dst = (DST);                                                                                                                 \
        for(int __i = (LANE_ID); __i < ((N) / kLoopStride) * kLoopStride; __i += kLoopStride) {                                             \
            _Pragma("unroll") for(int __j = 0; __j < (UNROLL_FACTOR); ++__j) unrolled_values[__j] = LD_FUNC(__src + __i + __j * kWarpSize); \
            _Pragma("unroll") for(int __j = 0; __j < (UNROLL_FACTOR); ++__j) ST_FUNC(__dst + __i + __j * kWarpSize, unrolled_values[__j]);  \
        }                                                                                                                                   \
        for(int __i = ((N) / kLoopStride) * kLoopStride + (LANE_ID); __i < (N); __i += kWarpSize)                                           \
            ST_FUNC(__dst + __i, LD_FUNC(__src + __i));                                                                                     \
    }


lijian6's avatar
lijian6 committed
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
#define UNROLLED_WARP_COPY_EMULATED(UNROLL_FACTOR, LANE_ID, N, DST, SRC, LD_FUNC, ST_FUNC)         \
    {                                                                                              \
        constexpr int kLoopStride = kEmulatedWarpSize * (UNROLL_FACTOR);                           \
        typename std::remove_reference<decltype(LD_FUNC((SRC) + 0))>::type                         \
             unrolled_values[(UNROLL_FACTOR)];                                                     \
        auto __src = (SRC);                                                                        \
        auto __dst = (DST);                                                                        \
        for (int __i = (LANE_ID); __i < ((N) / kLoopStride) * kLoopStride; __i += kLoopStride) {   \
            _Pragma("unroll") for (int __j = 0; __j < (UNROLL_FACTOR); ++__j)                      \
                unrolled_values[__j] = LD_FUNC(__src + __i + __j * kEmulatedWarpSize);             \
            _Pragma("unroll") for (int __j = 0; __j < (UNROLL_FACTOR); ++__j)                      \
                ST_FUNC(__dst + __i + __j * kEmulatedWarpSize, unrolled_values[__j]);              \
        }                                                                                          \
        for (int __i = ((N) / kLoopStride) * kLoopStride + (LANE_ID); __i < (N);                   \
             __i += kEmulatedWarpSize)                                                             \
            ST_FUNC(__dst + __i, LD_FUNC(__src + __i));                                            \
    }
// HELPER FUNCTIONS
// #####################################################################################

template <typename T>
__device__ __forceinline__ T shfl_xor(const T val, int laneMask, int width = kWarpSize,
                                      uint64_t shfl_sync_mask = kFullWarpMask) {
    return __shfl_xor(val, laneMask, width);
Chenggang Zhao's avatar
Chenggang Zhao committed
62
63
}

64
65
66
template <typename T>
__device__ __forceinline__ T shfl_sync(const T val, int srcLane = 0, int width = kWarpSize,
                                       uint64_t shfl_sync_mask = kFullWarpMask) { // Let compiler deduce type
lijian6's avatar
lijian6 committed
67
68
69
70
71
72
73
    return __shfl(val, srcLane, width);
}

__device__ __forceinline__ int __any_sync(uint64_t mask, int predicate) {
    uint64_t predicate_bit_pattern = __ballot(predicate);
    return (predicate_bit_pattern & mask) > 0;
}
Chenggang Zhao's avatar
Chenggang Zhao committed
74

lijian6's avatar
lijian6 committed
75
76
77
78
__device__ __forceinline__ int __all_sync(uint64_t mask, int predicate) {
    uint64_t predicate_bit_pattern = __ballot(predicate);
    return (~predicate_bit_pattern & mask) == 0;
}
Chenggang Zhao's avatar
Chenggang Zhao committed
79

lijian6's avatar
lijian6 committed
80
81
82
83
84
85
__device__ __forceinline__ void syncwarp() {
    __builtin_amdgcn_fence(__ATOMIC_RELEASE, "wavefront");
    __builtin_amdgcn_wave_barrier();
    __builtin_amdgcn_fence(__ATOMIC_ACQUIRE, "wavefront");
}
// ######################################################################################################
86

lijian6's avatar
lijian6 committed
87
namespace deep_ep {
88

lijian6's avatar
lijian6 committed
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
template <int kBytes> struct VecInt {};
template <> struct VecInt<1> {
    using vec_t = int8_t;
};
template <> struct VecInt<2> {
    using vec_t = int16_t;
};
template <> struct VecInt<4> {
    using vec_t = int;
};
template <> struct VecInt<8> {
    using vec_t = int64_t;
};
template <> struct VecInt<16> {
    using native_int4 = int __attribute__((ext_vector_type(4)));
    using vec_t       = native_int4;
105
106
};

107
108
109
110
111
112
113
114
115
template <typename FuncT>
struct PatternVisitor {
    FuncT func;

    __device__ __host__ explicit PatternVisitor(FuncT&& func) : func(std::forward<FuncT>(func)) {}

    __device__ __host__ auto operator[](const uint32_t& i) { return func(i); }
};

Chenggang Zhao's avatar
Chenggang Zhao committed
116
__device__ __forceinline__ void trap() {
lijian6's avatar
lijian6 committed
117
    abort();
Chenggang Zhao's avatar
Chenggang Zhao committed
118
119
120
}

__device__ __forceinline__ void memory_fence() {
lijian6's avatar
lijian6 committed
121
122

    __threadfence_system();
Chenggang Zhao's avatar
Chenggang Zhao committed
123
124
125
}

__device__ __forceinline__ void memory_fence_gpu() {
lijian6's avatar
lijian6 committed
126
    __threadfence();
Chenggang Zhao's avatar
Chenggang Zhao committed
127
128
129
}

__device__ __forceinline__ void memory_fence_cta() {
lijian6's avatar
lijian6 committed
130
    __threadfence_block();
Chenggang Zhao's avatar
Chenggang Zhao committed
131
132
}

lijian6's avatar
lijian6 committed
133
__device__ __forceinline__ void st_relaxed_sys_global(int *ptr, int val) {
lijian6's avatar
lijian6 committed
134
    __hip_atomic_store(ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
135
136
}

lijian6's avatar
lijian6 committed
137
138
__device__ __forceinline__ void st_release_sys_global(const int *ptr, int val) {
    __hip_atomic_store(const_cast<int *>(ptr), val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
139
140
}

141
142
143
144
__device__ __forceinline__ void st_release_sys_global(const int64_t *ptr, int64_t val) {
    __hip_atomic_store(const_cast<int64_t *>(ptr), val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_SYSTEM);
}

lijian6's avatar
lijian6 committed
145
146
147
148
149
__device__ __forceinline__ void st_release_cta(const int *ptr, int val) {
    __hip_atomic_store(const_cast<int *>(ptr), val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_WORKGROUP);
}

__device__ __forceinline__ int ld_relaxed_sys_global(const int *ptr) {
lijian6's avatar
lijian6 committed
150
151
152
    int ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
    return ret;
lijian6's avatar
lijian6 committed
153
154
155
156
157
}
__device__ __forceinline__ int ld_relaxed_sys_global(const uint64_t *ptr) {
    uint64_t ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
    return ret;
Chenggang Zhao's avatar
Chenggang Zhao committed
158
}
lijian6's avatar
lijian6 committed
159
160
161
162
163
__device__ __forceinline__ int ld_relaxed_sys_global(const int64_t *ptr) {
    int64_t ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
    return ret;
}
Chenggang Zhao's avatar
Chenggang Zhao committed
164
165
166

__device__ __forceinline__ int ld_acquire_sys_global(const int *ptr) {
    int ret;
lijian6's avatar
lijian6 committed
167
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
168
169
170
171
172
    return ret;
}

__device__ __forceinline__ uint64_t ld_acquire_sys_global(const uint64_t *ptr) {
    uint64_t ret;
lijian6's avatar
lijian6 committed
173
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
174
175
176
    return ret;
}

lijian6's avatar
lijian6 committed
177
178
179
180
181
182
183
__device__ __forceinline__ int64_t ld_acquire_sys_global(const int64_t *ptr) {
    int64_t ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_SYSTEM);
    return ret;
}


Chenggang Zhao's avatar
Chenggang Zhao committed
184
185
__device__ __forceinline__ int ld_acquire_global(const int *ptr) {
    int ret;
lijian6's avatar
lijian6 committed
186
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
187
188
189
    return ret;
}

190
191
192
193
194
195
__device__ __forceinline__ int64_t ld_acquire_global(const int64_t *ptr) {
    int64_t ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_AGENT);
    return ret;
}

lijian6's avatar
lijian6 committed
196
__device__ __forceinline__ int atomic_add_release_global(const int *ptr, int value) {
Chenggang Zhao's avatar
Chenggang Zhao committed
197
    int ret;
lishen's avatar
lishen committed
198
199
    ret = __hip_atomic_fetch_add(const_cast<int *>(ptr), value, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
    // ret = atomicAdd((int*)ptr, value);
Chenggang Zhao's avatar
Chenggang Zhao committed
200
201
202
    return ret;
}

203
204
205
206
207
208
__device__ __forceinline__ int ld_relaxed_global(const int *ptr) {
    int ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
    return ret;
}

Chenggang Zhao's avatar
Chenggang Zhao committed
209
210
__device__ __forceinline__ int ld_acquire_cta(const int *ptr) {
    int ret;
lijian6's avatar
lijian6 committed
211
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_WORKGROUP);
Chenggang Zhao's avatar
Chenggang Zhao committed
212
213
214
    return ret;
}

lijian6's avatar
lijian6 committed
215
__device__ __forceinline__ int ld_volatile_global(const volatile int *ptr) {
Chenggang Zhao's avatar
Chenggang Zhao committed
216
    int ret;
lijian6's avatar
lijian6 committed
217
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
218
219
220
    return ret;
}

lijian6's avatar
lijian6 committed
221
__device__ __forceinline__ float ld_volatile_global(const volatile float *ptr) {
Chenggang Zhao's avatar
Chenggang Zhao committed
222
    float ret;
lijian6's avatar
lijian6 committed
223
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
224
225
226
    return ret;
}

lijian6's avatar
lijian6 committed
227
__device__ __forceinline__ int64_t ld_volatile_global(const volatile int64_t *ptr) {
Chenggang Zhao's avatar
Chenggang Zhao committed
228
    int64_t ret;
lijian6's avatar
lijian6 committed
229
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
230
231
232
    return ret;
}

lijian6's avatar
lijian6 committed
233
__device__ __forceinline__ int64_t ld_volatile_global(const volatile uint64_t *ptr) {
Chenggang Zhao's avatar
Chenggang Zhao committed
234
    int64_t ret;
lijian6's avatar
lijian6 committed
235
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
236
237
238
    return ret;
}

lijian6's avatar
lijian6 committed
239
template <typename dtype_t> __device__ __forceinline__ dtype_t ld_nc_global(const dtype_t *ptr) {
lijian6's avatar
lijian6 committed
240
241
242
    using T  = typename VecInt<sizeof(dtype_t)>::vec_t;
    auto ret = __builtin_nontemporal_load(reinterpret_cast<const T *>(ptr));
    return *reinterpret_cast<dtype_t *>(&ret);
Chenggang Zhao's avatar
Chenggang Zhao committed
243
244
}

lishen's avatar
lishen committed
245
246
247
248
249
250
template <typename dtype_t> __device__ __forceinline__ dtype_t ld_direct_global(const dtype_t *ptr) {
    using T  = typename VecInt<sizeof(dtype_t)>::vec_t;
    auto ret = *(reinterpret_cast<const T *>(ptr));
    return *reinterpret_cast<dtype_t *>(&ret);
}

lijian6's avatar
lijian6 committed
251
////////////////// used in ibgda
Chenggang Zhao's avatar
Chenggang Zhao committed
252
__device__ __forceinline__ void st_na_relaxed(const uint8_t *ptr, uint8_t val) {
lijian6's avatar
lijian6 committed
253
254
    uint8_t *non_const_ptr = const_cast<uint8_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
255
256
257
}

__device__ __forceinline__ void st_na_relaxed(const uint16_t *ptr, uint16_t val) {
lijian6's avatar
lijian6 committed
258
259
    uint16_t *non_const_ptr = const_cast<uint16_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
260
261
262
}

__device__ __forceinline__ void st_na_relaxed(const uint32_t *ptr, uint32_t val) {
lijian6's avatar
lijian6 committed
263
264
    uint32_t *non_const_ptr = const_cast<uint32_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
265
266
267
}

__device__ __forceinline__ void st_na_relaxed(const int *ptr, int val) {
lijian6's avatar
lijian6 committed
268
269
    int *non_const_ptr = const_cast<int *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
270
271
272
}

__device__ __forceinline__ void st_na_relaxed(const int4 *ptr, int4 val) {
lijian6's avatar
lijian6 committed
273
274
275
276
277
    int4 *non_const_ptr = const_cast<int4 *>(ptr);
    non_const_ptr->x    = val.x;
    non_const_ptr->y    = val.y;
    non_const_ptr->z    = val.z;
    non_const_ptr->w    = val.w;
Chenggang Zhao's avatar
Chenggang Zhao committed
278
279
280
}

__device__ __forceinline__ void st_na_release(const int *ptr, int val) {
lijian6's avatar
lijian6 committed
281
282
    int *non_const_ptr = const_cast<int *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
283
284
285
}

__device__ __forceinline__ void st_na_release(const uint32_t *ptr, uint32_t val) {
lijian6's avatar
lijian6 committed
286
287
    uint32_t *non_const_ptr = const_cast<uint32_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
288
289
290
}

__device__ __forceinline__ void st_na_release(const uint64_t *ptr, uint64_t val) {
lijian6's avatar
lijian6 committed
291
292
    uint64_t *non_const_ptr = const_cast<uint64_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
293
294
}

295
296
297
298
299
__device__ __forceinline__ void st_na_release(const int64_t *ptr, int64_t val) {
    int64_t *non_const_ptr = const_cast<int64_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
}

lijian6's avatar
lijian6 committed
300
// TODO:: apply "st.global.L1::no_allocate" in ROCM
Chenggang Zhao's avatar
Chenggang Zhao committed
301
template <typename dtype_t>
lijian6's avatar
lijian6 committed
302
303
304
__device__ __forceinline__ void st_na_global(const dtype_t *ptr, const dtype_t &value) {
    st_na_global(reinterpret_cast<const typename VecInt<sizeof(dtype_t)>::vec_t *>(ptr),
                 *reinterpret_cast<const typename VecInt<sizeof(dtype_t)>::vec_t *>(&value));
Chenggang Zhao's avatar
Chenggang Zhao committed
305
306
}

lijian6's avatar
lijian6 committed
307
308
309
template <> __device__ __forceinline__ void st_na_global(const int *ptr, const int &value) {
    int *non_const_ptr = const_cast<int *>(ptr);
    *non_const_ptr     = value;
310
311
}

lijian6's avatar
lijian6 committed
312
313
314
template <> __device__ __forceinline__ void st_na_global(const int64_t *ptr, const int64_t &value) {
    int64_t *non_const_ptr = const_cast<int64_t *>(ptr);
    *non_const_ptr         = value;
Chenggang Zhao's avatar
Chenggang Zhao committed
315
316
}

lijian6's avatar
lijian6 committed
317
318
319
template <> __device__ __forceinline__ void st_na_global(const float *ptr, const float &value) {
    float *non_const_ptr = const_cast<float *>(ptr);
    *non_const_ptr       = value;
Chenggang Zhao's avatar
Chenggang Zhao committed
320
321
}

lijian6's avatar
lijian6 committed
322
323
324
template <> __device__ __forceinline__ void st_na_global(const int4 *ptr, const int4 &value) {
    int4 *non_const_ptr = const_cast<int4 *>(ptr);
    *non_const_ptr      = value;
325
326
}

Chenggang Zhao's avatar
Chenggang Zhao committed
327
__forceinline__ __device__ void get_channel_task_range(int num_tokens, int num_sms, int sm_id,
lijian6's avatar
lijian6 committed
328
329
330
331
                                                       int &token_start_idx, int &token_end_idx) {
    int num_tokens_per_sm = DIVUP(num_tokens, num_sms);
    token_start_idx       = min(num_tokens_per_sm * sm_id, num_tokens);
    token_end_idx         = min(token_start_idx + num_tokens_per_sm, num_tokens);
Chenggang Zhao's avatar
Chenggang Zhao committed
332
333
}

334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
template <typename dtype_a_t, typename dtype_b_t>
__device__ __forceinline__ dtype_b_t pack2(const dtype_a_t& x, const dtype_a_t& y) {
    EP_STATIC_ASSERT(sizeof(dtype_a_t) * 2 == sizeof(dtype_b_t), "Invalid dtypes");
    dtype_b_t packed;
    auto unpacked_ptr = reinterpret_cast<dtype_a_t*>(&packed);
    unpacked_ptr[0] = x, unpacked_ptr[1] = y;
    return packed;
}

template <typename dtype_a_t, typename dtype_b_t>
__device__ __forceinline__ void unpack2(const dtype_b_t& packed, dtype_a_t& x, dtype_a_t& y) {
    EP_STATIC_ASSERT(sizeof(dtype_a_t) * 2 == sizeof(dtype_b_t), "Invalid dtypes");
    auto unpacked_ptr = reinterpret_cast<const dtype_a_t*>(&packed);
    x = unpacked_ptr[0], y = unpacked_ptr[1];
}

Chenggang Zhao's avatar
Chenggang Zhao committed
350
template <typename dtype_t>
lijian6's avatar
lijian6 committed
351
__device__ __forceinline__ dtype_t broadcast(dtype_t &ptr, int src_lane_idx) {
Chenggang Zhao's avatar
Chenggang Zhao committed
352
    EP_STATIC_ASSERT(sizeof(dtype_t) % sizeof(int) == 0, "");
lijian6's avatar
lijian6 committed
353
354
355
356
357
358
359
360
    auto send_int_values = reinterpret_cast<int *>(&ptr);
    int  recv_int_values[sizeof(dtype_t) / sizeof(int)];
#pragma unroll
    for (int i = 0; i < sizeof(dtype_t) / sizeof(int); ++i)
        recv_int_values[i] = shfl_sync(send_int_values[i], src_lane_idx);
    return *reinterpret_cast<dtype_t *>(recv_int_values);
}

361
// 设置不同的量化方式的最大值与相反数
lishen's avatar
lishen committed
362
constexpr float kFinfoAmaxE4M3 = 448.0f;
lishen's avatar
lishen committed
363
constexpr float kFinfoAmaxInvE4M3 = 1.0f / kFinfoAmaxE4M3;
364
365
366
367
constexpr float kFinfoAmaxE5M2 = 57344.0f; 
constexpr float kFinfoAmaxInvE5M2 = 1.0f / kFinfoAmaxE5M2;
constexpr float kFinfoAmaxInt8 = 127.0f;
constexpr float kFinfoAmaxInvInt8 = 1.0f / 127.0f;
368
369
370
371
372
373
374
375
376
377
378
379
380
381

__forceinline__ __device__ float fast_pow2(int x) {
    // We can ensure `-126 <= x and x <= 127`
    uint32_t bits_x = (x + 127) << 23;
    return *reinterpret_cast<float*>(&bits_x);
}

__forceinline__ __device__ int fast_log2_ceil(float x) {
    auto bits_x = *reinterpret_cast<uint32_t*>(&x);
    auto exp_x = (bits_x >> 23) & 0xff;
    auto man_bits = bits_x & ((1 << 23) - 1);
    return exp_x - 127 + (man_bits != 0);
}

382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
template <int kQuantType>
__forceinline__ __device__ void calculate_quant8bit_scales(float amax, float& scale, float& scale_inv, bool round_scale=0) {
    amax = fmaxf(amax, 1e-6f);
    if constexpr(kQuantType == 1) { // 使用 INT8 对称量化
        scale_inv = kFinfoAmaxInvInt8 * amax;
        scale = kFinfoAmaxInt8 / amax;
    } else if constexpr(kQuantType == 2 || kQuantType == 3) {   // 使用 FP8_E4M3 或 FP8_UE8M0 非对称量化
        if (round_scale) {
            auto exp_scale_inv = fast_log2_ceil(amax * kFinfoAmaxInvE4M3);
            scale = fast_pow2(-exp_scale_inv);
            scale_inv = fast_pow2(exp_scale_inv);
        } else {
            scale_inv = amax * kFinfoAmaxInvE4M3;
            scale = kFinfoAmaxE4M3 / amax;
        }
    } else if constexpr(kQuantType == 4) { // 使用 FP8_E5M2 对称量化
        if (round_scale) {
            auto exp_scale_inv = fast_log2_ceil(amax * kFinfoAmaxInvE5M2);
            scale = fast_pow2(-exp_scale_inv);
            scale_inv = fast_pow2(exp_scale_inv);
        } else {
            scale_inv = amax * kFinfoAmaxInvE5M2;
            scale = kFinfoAmaxE5M2 / amax;
        }
406
407
408
409
410
411
412
413
414
415
    }
}

template <bool kIsUE8M0, typename out_dtype_t = std::conditional_t<kIsUE8M0, uint8_t, float>>
__forceinline__ __device__ out_dtype_t extract_required_scale_format(float value) {
    if constexpr (kIsUE8M0) {
        return static_cast<uint8_t>((*reinterpret_cast<uint32_t*>(&value)) >> 23);
    } else {
        return value;
    }
Shifang Xu's avatar
Shifang Xu committed
416
417
}

lijian6's avatar
lijian6 committed
418
419
420
__forceinline__ __device__ int get_lane_id() {
    int lane_id = threadIdx.x % kWarpSize;
    return lane_id;
Shifang Xu's avatar
Shifang Xu committed
421
422
}

423
template <int kNumRanks, bool kSyncOnly = false>
lijian6's avatar
lijian6 committed
424
__forceinline__ __device__ void barrier_block(int **barrier_signal_ptrs, int rank) {
Chenggang Zhao's avatar
Chenggang Zhao committed
425
426
    auto thread_id = static_cast<int>(threadIdx.x);

lijian6's avatar
lijian6 committed
427
428
    // For non-sync-only cases, the memory operations by other threads in the block must be visible
    // to the `sys` scope
429
430
431
432
433
    if constexpr (not kSyncOnly) {
        memory_fence();
        __syncthreads();
    }

434
    // Add self-ranks, sub other ranks
Chenggang Zhao's avatar
Chenggang Zhao committed
435
    if (thread_id < kNumRanks) {
436
437
438
439
440
441
442
443
        atomicAdd_system(barrier_signal_ptrs[rank] + thread_id, FINISHED_SUM_TAG);
        atomicSub_system(barrier_signal_ptrs[thread_id] + rank, FINISHED_SUM_TAG);
    }
    EP_DEVICE_ASSERT(kNumRanks <= blockDim.x);

    // Check timeout
    auto start_time = clock64();
    while (true) {
lijian6's avatar
lijian6 committed
444
445
446
        auto value =
            thread_id < kNumRanks ? ld_volatile_global(barrier_signal_ptrs[rank] + thread_id) : 0;
        if (__all_sync(kFullWarpMask, value <= 0))
447
448
            break;

Chenggang Zhao's avatar
Chenggang Zhao committed
449
        if (clock64() - start_time > NUM_TIMEOUT_CYCLES and thread_id < kNumRanks) {
lijian6's avatar
lijian6 committed
450
451
            printf("DeepEP timeout check failed: rank = %d, thread = %d, value = %d)\n", rank,
                   thread_id, value);
452
453
            trap();
        }
Chenggang Zhao's avatar
Chenggang Zhao committed
454
    }
455
    __syncthreads();
Chenggang Zhao's avatar
Chenggang Zhao committed
456
}
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547

// Operation functors
template <typename T>
struct ReduceSum {
    __device__ T operator()(T a, T b) const { return a + b; }
};
template <typename T>
struct ReduceMax {
    __device__ T operator()(T a, T b) const { return a > b ? a : b; }
};
template <typename T>
struct ReduceMin {
    __device__ T operator()(T a, T b) const { return a < b ? a : b; }
};
template <typename T>
struct ReduceAnd {
    __device__ T operator()(T a, T b) const { return a & b; }
};
template <typename T>
struct ReduceOr {
    __device__ T operator()(T a, T b) const { return a | b; }
};

// Unified reduction function
template <int kNumLanesPerGroup, bool kIntergroupReduce, typename T, typename Op>
__forceinline__ __device__ T warp_reduce(T value, Op op) {
    EP_STATIC_ASSERT(kNumLanesPerGroup == kWarpSize or kNumLanesPerGroup == 32 or
                     kNumLanesPerGroup == 16 or kNumLanesPerGroup == 8 or kNumLanesPerGroup == 4 or
                     kNumLanesPerGroup == 2 or kNumLanesPerGroup == 1,
                     "Invalid number of lanes");
    constexpr uint32_t mask = 0xffffffff;
    if constexpr (kIntergroupReduce) {
        if constexpr (kNumLanesPerGroup <= 1)
        value = op(value, shfl_xor(value, 1));
        if constexpr (kNumLanesPerGroup <= 2)
        value = op(value, shfl_xor(value, 2));
        if constexpr (kNumLanesPerGroup <= 4)
        value = op(value, shfl_xor(value, 4));
        if constexpr (kNumLanesPerGroup <= 8)
        value = op(value, shfl_xor(value, 8));
        if constexpr (kNumLanesPerGroup <= 16)
        value = op(value, shfl_xor(value, 16));
        if constexpr(kWarpSize == 64){
            if constexpr (kNumLanesPerGroup <= 32)
            value = op(value, shfl_xor(value, 32));
        }
    } else {
        if constexpr(kWarpSize == 64){
            if constexpr (kNumLanesPerGroup >= kWarpSize)
            value = op(value, shfl_xor(value, 32));
        }
        if constexpr (kNumLanesPerGroup >= 32)
        value = op(value, shfl_xor(value, 16));
        if constexpr (kNumLanesPerGroup >= 16)
        value = op(value, shfl_xor(value, 8));
        if constexpr (kNumLanesPerGroup >= 8)
        value = op(value, shfl_xor(value, 4));
        if constexpr (kNumLanesPerGroup >= 4)
        value = op(value, shfl_xor(value, 2));
        if constexpr (kNumLanesPerGroup >= 2)
        value = op(value, shfl_xor(value, 1));
    }
    return value;
}

// Convenience aliases
template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_sum(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceSum<T>{});
}

template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_max(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceMax<T>{});
}

template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_min(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceMin<T>{});
}

template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_and(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceAnd<T>{});
}

template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_or(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceOr<T>{});
}

Chenggang Zhao's avatar
Chenggang Zhao committed
548
} // namespace deep_ep