utils.cuh 22.3 KB
Newer Older
lijian6's avatar
lijian6 committed
1
#include "hip/hip_runtime.h"
Chenggang Zhao's avatar
Chenggang Zhao committed
2
#pragma once
lijian6's avatar
lijian6 committed
3
#include "configs.cuh"
Chenggang Zhao's avatar
Chenggang Zhao committed
4
5
#include "exception.cuh"

lijian6's avatar
lijian6 committed
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
#define UNROLLED_WARP_COPY(UNROLL_FACTOR, LANE_ID, N, DST, SRC, LD_FUNC, ST_FUNC)                  \
    {                                                                                              \
        constexpr int kLoopStride = kWarpSize * (UNROLL_FACTOR);                                   \
        typename std::remove_reference<decltype(LD_FUNC((SRC) + 0))>::type                         \
             unrolled_values[(UNROLL_FACTOR)];                                                     \
        auto __src = (SRC);                                                                        \
        auto __dst = (DST);                                                                        \
        for (int __i = (LANE_ID); __i < ((N) / kLoopStride) * kLoopStride; __i += kLoopStride) {   \
            _Pragma("unroll") for (int __j = 0; __j < (UNROLL_FACTOR); ++__j)                      \
                unrolled_values[__j] = LD_FUNC(__src + __i + __j * kWarpSize);                     \
            _Pragma("unroll") for (int __j = 0; __j < (UNROLL_FACTOR); ++__j)                      \
                ST_FUNC(__dst + __i + __j * kWarpSize, unrolled_values[__j]);                      \
        }                                                                                          \
        {                                                                                          \
            int __i = ((N) / kLoopStride) * kLoopStride + (LANE_ID);                               \
            _Pragma("unroll") for (int __j = 0; __j < (UNROLL_FACTOR); ++__j) {                    \
                if (__i + __j * kWarpSize < (N)) {                                                 \
                    unrolled_values[__j] = LD_FUNC(__src + __i + __j * kWarpSize);                 \
                }                                                                                  \
            }                                                                                      \
            _Pragma("unroll") for (int __j = 0; __j < (UNROLL_FACTOR); ++__j) {                    \
                if (__i + __j * kWarpSize < (N)) {                                                 \
                    ST_FUNC(__dst + __i + __j * kWarpSize, unrolled_values[__j]);                  \
                }                                                                                  \
            }                                                                                      \
        }                                                                                          \
    }

lishen's avatar
lishen committed
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
#define UNROLLED_WARP_COPY_LL(UNROLL_FACTOR, LANE_ID, N, DST, SRC, LD_FUNC, ST_FUNC)                                                        \
    {                                                                                                                                       \
        constexpr int kLoopStride = kWarpSize * (UNROLL_FACTOR);                                                                            \
        typename std::remove_reference<decltype(LD_FUNC((SRC) + 0))>::type unrolled_values[(UNROLL_FACTOR)];                                \
        auto __src = (SRC);                                                                                                                 \
        auto __dst = (DST);                                                                                                                 \
        for(int __i = (LANE_ID); __i < ((N) / kLoopStride) * kLoopStride; __i += kLoopStride) {                                             \
            _Pragma("unroll") for(int __j = 0; __j < (UNROLL_FACTOR); ++__j) unrolled_values[__j] = LD_FUNC(__src + __i + __j * kWarpSize); \
            _Pragma("unroll") for(int __j = 0; __j < (UNROLL_FACTOR); ++__j) ST_FUNC(__dst + __i + __j * kWarpSize, unrolled_values[__j]);  \
        }                                                                                                                                   \
        for(int __i = ((N) / kLoopStride) * kLoopStride + (LANE_ID); __i < (N); __i += kWarpSize)                                           \
            ST_FUNC(__dst + __i, LD_FUNC(__src + __i));                                                                                     \
    }


lijian6's avatar
lijian6 committed
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
#define UNROLLED_WARP_COPY_EMULATED(UNROLL_FACTOR, LANE_ID, N, DST, SRC, LD_FUNC, ST_FUNC)         \
    {                                                                                              \
        constexpr int kLoopStride = kEmulatedWarpSize * (UNROLL_FACTOR);                           \
        typename std::remove_reference<decltype(LD_FUNC((SRC) + 0))>::type                         \
             unrolled_values[(UNROLL_FACTOR)];                                                     \
        auto __src = (SRC);                                                                        \
        auto __dst = (DST);                                                                        \
        for (int __i = (LANE_ID); __i < ((N) / kLoopStride) * kLoopStride; __i += kLoopStride) {   \
            _Pragma("unroll") for (int __j = 0; __j < (UNROLL_FACTOR); ++__j)                      \
                unrolled_values[__j] = LD_FUNC(__src + __i + __j * kEmulatedWarpSize);             \
            _Pragma("unroll") for (int __j = 0; __j < (UNROLL_FACTOR); ++__j)                      \
                ST_FUNC(__dst + __i + __j * kEmulatedWarpSize, unrolled_values[__j]);              \
        }                                                                                          \
        for (int __i = ((N) / kLoopStride) * kLoopStride + (LANE_ID); __i < (N);                   \
             __i += kEmulatedWarpSize)                                                             \
            ST_FUNC(__dst + __i, LD_FUNC(__src + __i));                                            \
    }
// HELPER FUNCTIONS
// #####################################################################################

template <typename T>
__device__ __forceinline__ T shfl_xor(const T val, int laneMask, int width = kWarpSize,
                                      uint64_t shfl_sync_mask = kFullWarpMask) {
    return __shfl_xor(val, laneMask, width);
Chenggang Zhao's avatar
Chenggang Zhao committed
73
74
}

lijian6's avatar
lijian6 committed
75
76
77
78
79
80
81
82
83
84
__device__ __forceinline__ int
shfl_sync(const int val, int srcLane = 0, int width = kWarpSize,
          uint64_t shfl_sync_mask = kFullWarpMask) { // Let compiler deduce type
    return __shfl(val, srcLane, width);
}

__device__ __forceinline__ int __any_sync(uint64_t mask, int predicate) {
    uint64_t predicate_bit_pattern = __ballot(predicate);
    return (predicate_bit_pattern & mask) > 0;
}
Chenggang Zhao's avatar
Chenggang Zhao committed
85

lijian6's avatar
lijian6 committed
86
87
88
89
__device__ __forceinline__ int __all_sync(uint64_t mask, int predicate) {
    uint64_t predicate_bit_pattern = __ballot(predicate);
    return (~predicate_bit_pattern & mask) == 0;
}
Chenggang Zhao's avatar
Chenggang Zhao committed
90

lijian6's avatar
lijian6 committed
91
92
93
94
95
96
__device__ __forceinline__ void syncwarp() {
    __builtin_amdgcn_fence(__ATOMIC_RELEASE, "wavefront");
    __builtin_amdgcn_wave_barrier();
    __builtin_amdgcn_fence(__ATOMIC_ACQUIRE, "wavefront");
}
// ######################################################################################################
97

lijian6's avatar
lijian6 committed
98
namespace deep_ep {
99

lijian6's avatar
lijian6 committed
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
template <int kBytes> struct VecInt {};
template <> struct VecInt<1> {
    using vec_t = int8_t;
};
template <> struct VecInt<2> {
    using vec_t = int16_t;
};
template <> struct VecInt<4> {
    using vec_t = int;
};
template <> struct VecInt<8> {
    using vec_t = int64_t;
};
template <> struct VecInt<16> {
    using native_int4 = int __attribute__((ext_vector_type(4)));
    using vec_t       = native_int4;
116
117
};

Chenggang Zhao's avatar
Chenggang Zhao committed
118
__device__ __forceinline__ void trap() {
lijian6's avatar
lijian6 committed
119
    abort();
Chenggang Zhao's avatar
Chenggang Zhao committed
120
121
122
}

__device__ __forceinline__ void memory_fence() {
lijian6's avatar
lijian6 committed
123
124

    __threadfence_system();
Chenggang Zhao's avatar
Chenggang Zhao committed
125
126
127
}

__device__ __forceinline__ void memory_fence_gpu() {
lijian6's avatar
lijian6 committed
128
    __threadfence();
Chenggang Zhao's avatar
Chenggang Zhao committed
129
130
131
}

__device__ __forceinline__ void memory_fence_cta() {
lijian6's avatar
lijian6 committed
132
    __threadfence_block();
Chenggang Zhao's avatar
Chenggang Zhao committed
133
134
}

lijian6's avatar
lijian6 committed
135
136
__device__ __forceinline__ void st_relaxed_sys_global(int *ptr, int val) {
    __builtin_nontemporal_store(val, ptr);
Chenggang Zhao's avatar
Chenggang Zhao committed
137
138
}

lijian6's avatar
lijian6 committed
139
140
__device__ __forceinline__ void st_release_sys_global(const int *ptr, int val) {
    __hip_atomic_store(const_cast<int *>(ptr), val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
141
142
}

143
144
145
146
__device__ __forceinline__ void st_release_sys_global(const int64_t *ptr, int64_t val) {
    __hip_atomic_store(const_cast<int64_t *>(ptr), val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_SYSTEM);
}

lijian6's avatar
lijian6 committed
147
148
149
150
151
152
153
154
155
156
157
158
__device__ __forceinline__ void st_release_cta(const int *ptr, int val) {
    __hip_atomic_store(const_cast<int *>(ptr), val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_WORKGROUP);
}

__device__ __forceinline__ int ld_relaxed_sys_global(const int *ptr) {
    int res = __builtin_nontemporal_load(ptr);
    return res;
}
__device__ __forceinline__ int ld_relaxed_sys_global(const uint64_t *ptr) {
    uint64_t ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
    return ret;
Chenggang Zhao's avatar
Chenggang Zhao committed
159
160
161
162
}

__device__ __forceinline__ int ld_acquire_sys_global(const int *ptr) {
    int ret;
lijian6's avatar
lijian6 committed
163
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
164
165
166
167
168
    return ret;
}

__device__ __forceinline__ uint64_t ld_acquire_sys_global(const uint64_t *ptr) {
    uint64_t ret;
lijian6's avatar
lijian6 committed
169
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
170
171
172
173
174
    return ret;
}

__device__ __forceinline__ int ld_acquire_global(const int *ptr) {
    int ret;
lijian6's avatar
lijian6 committed
175
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
176
177
178
    return ret;
}

179
180
181
182
183
184
__device__ __forceinline__ int64_t ld_acquire_global(const int64_t *ptr) {
    int64_t ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_AGENT);
    return ret;
}

lijian6's avatar
lijian6 committed
185
__device__ __forceinline__ int atomic_add_release_global(const int *ptr, int value) {
Chenggang Zhao's avatar
Chenggang Zhao committed
186
    int ret;
lijian6's avatar
lijian6 committed
187
188
189
    // ret = __hip_atomic_fetch_add(const_cast<int *>(ptr), value, __ATOMIC_RELEASE,
    //                              __HIP_MEMORY_SCOPE_AGENT);
    ret = atomicAdd((int*)ptr, value);
Chenggang Zhao's avatar
Chenggang Zhao committed
190
191
192
    return ret;
}

193
194
195
196
197
198
__device__ __forceinline__ int ld_relaxed_global(const int *ptr) {
    int ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
    return ret;
}

Chenggang Zhao's avatar
Chenggang Zhao committed
199
200
__device__ __forceinline__ int ld_acquire_cta(const int *ptr) {
    int ret;
lijian6's avatar
lijian6 committed
201
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_WORKGROUP);
Chenggang Zhao's avatar
Chenggang Zhao committed
202
203
204
    return ret;
}

lijian6's avatar
lijian6 committed
205
__device__ __forceinline__ int ld_volatile_global(const volatile int *ptr) {
Chenggang Zhao's avatar
Chenggang Zhao committed
206
    int ret;
lijian6's avatar
lijian6 committed
207
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
208
209
210
    return ret;
}

lijian6's avatar
lijian6 committed
211
__device__ __forceinline__ float ld_volatile_global(const volatile float *ptr) {
Chenggang Zhao's avatar
Chenggang Zhao committed
212
    float ret;
lijian6's avatar
lijian6 committed
213
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
214
215
216
    return ret;
}

lijian6's avatar
lijian6 committed
217
__device__ __forceinline__ int64_t ld_volatile_global(const volatile int64_t *ptr) {
Chenggang Zhao's avatar
Chenggang Zhao committed
218
    int64_t ret;
lijian6's avatar
lijian6 committed
219
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
220
221
222
    return ret;
}

lijian6's avatar
lijian6 committed
223
__device__ __forceinline__ int64_t ld_volatile_global(const volatile uint64_t *ptr) {
Chenggang Zhao's avatar
Chenggang Zhao committed
224
    int64_t ret;
lijian6's avatar
lijian6 committed
225
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
226
227
228
    return ret;
}

lijian6's avatar
lijian6 committed
229
template <typename dtype_t> __device__ __forceinline__ dtype_t ld_nc_global(const dtype_t *ptr) {
lijian6's avatar
lijian6 committed
230
231
232
    using T  = typename VecInt<sizeof(dtype_t)>::vec_t;
    auto ret = __builtin_nontemporal_load(reinterpret_cast<const T *>(ptr));
    return *reinterpret_cast<dtype_t *>(&ret);
Chenggang Zhao's avatar
Chenggang Zhao committed
233
234
}

lijian6's avatar
lijian6 committed
235
////////////////// used in ibgda
Chenggang Zhao's avatar
Chenggang Zhao committed
236
__device__ __forceinline__ void st_na_relaxed(const uint8_t *ptr, uint8_t val) {
lijian6's avatar
lijian6 committed
237
238
    uint8_t *non_const_ptr = const_cast<uint8_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
239
240
241
}

__device__ __forceinline__ void st_na_relaxed(const uint16_t *ptr, uint16_t val) {
lijian6's avatar
lijian6 committed
242
243
    uint16_t *non_const_ptr = const_cast<uint16_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
244
245
246
}

__device__ __forceinline__ void st_na_relaxed(const uint32_t *ptr, uint32_t val) {
lijian6's avatar
lijian6 committed
247
248
    uint32_t *non_const_ptr = const_cast<uint32_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
249
250
251
}

__device__ __forceinline__ void st_na_relaxed(const int *ptr, int val) {
lijian6's avatar
lijian6 committed
252
253
    int *non_const_ptr = const_cast<int *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
254
255
256
}

__device__ __forceinline__ void st_na_relaxed(const int4 *ptr, int4 val) {
lijian6's avatar
lijian6 committed
257
258
259
260
261
    int4 *non_const_ptr = const_cast<int4 *>(ptr);
    non_const_ptr->x    = val.x;
    non_const_ptr->y    = val.y;
    non_const_ptr->z    = val.z;
    non_const_ptr->w    = val.w;
Chenggang Zhao's avatar
Chenggang Zhao committed
262
263
264
}

__device__ __forceinline__ void st_na_release(const int *ptr, int val) {
lijian6's avatar
lijian6 committed
265
266
    int *non_const_ptr = const_cast<int *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
267
268
269
}

__device__ __forceinline__ void st_na_release(const uint32_t *ptr, uint32_t val) {
lijian6's avatar
lijian6 committed
270
271
    uint32_t *non_const_ptr = const_cast<uint32_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
272
273
274
}

__device__ __forceinline__ void st_na_release(const uint64_t *ptr, uint64_t val) {
lijian6's avatar
lijian6 committed
275
276
    uint64_t *non_const_ptr = const_cast<uint64_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
277
278
}

279
280
281
282
283
__device__ __forceinline__ void st_na_release(const int64_t *ptr, int64_t val) {
    int64_t *non_const_ptr = const_cast<int64_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
}

lijian6's avatar
lijian6 committed
284
// TODO:: apply "st.global.L1::no_allocate" in ROCM
Chenggang Zhao's avatar
Chenggang Zhao committed
285
template <typename dtype_t>
lijian6's avatar
lijian6 committed
286
287
288
__device__ __forceinline__ void st_na_global(const dtype_t *ptr, const dtype_t &value) {
    st_na_global(reinterpret_cast<const typename VecInt<sizeof(dtype_t)>::vec_t *>(ptr),
                 *reinterpret_cast<const typename VecInt<sizeof(dtype_t)>::vec_t *>(&value));
Chenggang Zhao's avatar
Chenggang Zhao committed
289
290
}

lijian6's avatar
lijian6 committed
291
292
293
template <> __device__ __forceinline__ void st_na_global(const int *ptr, const int &value) {
    int *non_const_ptr = const_cast<int *>(ptr);
    *non_const_ptr     = value;
294
295
}

lijian6's avatar
lijian6 committed
296
297
298
template <> __device__ __forceinline__ void st_na_global(const int64_t *ptr, const int64_t &value) {
    int64_t *non_const_ptr = const_cast<int64_t *>(ptr);
    *non_const_ptr         = value;
Chenggang Zhao's avatar
Chenggang Zhao committed
299
300
}

lijian6's avatar
lijian6 committed
301
302
303
template <> __device__ __forceinline__ void st_na_global(const float *ptr, const float &value) {
    float *non_const_ptr = const_cast<float *>(ptr);
    *non_const_ptr       = value;
Chenggang Zhao's avatar
Chenggang Zhao committed
304
305
}

lijian6's avatar
lijian6 committed
306
307
308
template <> __device__ __forceinline__ void st_na_global(const int4 *ptr, const int4 &value) {
    int4 *non_const_ptr = const_cast<int4 *>(ptr);
    *non_const_ptr      = value;
309
310
}

Chenggang Zhao's avatar
Chenggang Zhao committed
311
__forceinline__ __device__ void get_channel_task_range(int num_tokens, int num_sms, int sm_id,
lijian6's avatar
lijian6 committed
312
313
314
315
                                                       int &token_start_idx, int &token_end_idx) {
    int num_tokens_per_sm = DIVUP(num_tokens, num_sms);
    token_start_idx       = min(num_tokens_per_sm * sm_id, num_tokens);
    token_end_idx         = min(token_start_idx + num_tokens_per_sm, num_tokens);
Chenggang Zhao's avatar
Chenggang Zhao committed
316
317
}

318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
template <typename dtype_a_t, typename dtype_b_t>
__device__ __forceinline__ dtype_b_t pack2(const dtype_a_t& x, const dtype_a_t& y) {
    EP_STATIC_ASSERT(sizeof(dtype_a_t) * 2 == sizeof(dtype_b_t), "Invalid dtypes");
    dtype_b_t packed;
    auto unpacked_ptr = reinterpret_cast<dtype_a_t*>(&packed);
    unpacked_ptr[0] = x, unpacked_ptr[1] = y;
    return packed;
}

template <typename dtype_a_t, typename dtype_b_t>
__device__ __forceinline__ void unpack2(const dtype_b_t& packed, dtype_a_t& x, dtype_a_t& y) {
    EP_STATIC_ASSERT(sizeof(dtype_a_t) * 2 == sizeof(dtype_b_t), "Invalid dtypes");
    auto unpacked_ptr = reinterpret_cast<const dtype_a_t*>(&packed);
    x = unpacked_ptr[0], y = unpacked_ptr[1];
}

Chenggang Zhao's avatar
Chenggang Zhao committed
334
template <typename dtype_t>
lijian6's avatar
lijian6 committed
335
__device__ __forceinline__ dtype_t broadcast(dtype_t &ptr, int src_lane_idx) {
Chenggang Zhao's avatar
Chenggang Zhao committed
336
    EP_STATIC_ASSERT(sizeof(dtype_t) % sizeof(int) == 0, "");
lijian6's avatar
lijian6 committed
337
338
339
340
341
342
343
344
    auto send_int_values = reinterpret_cast<int *>(&ptr);
    int  recv_int_values[sizeof(dtype_t) / sizeof(int)];
#pragma unroll
    for (int i = 0; i < sizeof(dtype_t) / sizeof(int); ++i)
        recv_int_values[i] = shfl_sync(send_int_values[i], src_lane_idx);
    return *reinterpret_cast<dtype_t *>(recv_int_values);
}

345
346
#ifdef USE_ROCM
constexpr float kFP8Margin = 1e-4;
lishen's avatar
lishen committed
347
348
constexpr float kFinfoAmaxE4M3 = 240.0f;
constexpr float kFinfoAmaxInvE4M3 = 1.0f / kFinfoAmaxE4M3;
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
#else
constexpr float kFP8Margin = 1e-4;
constexpr float kFinfoAmaxE4M3 = 448.0f;
constexpr float kFinfoAmaxInvE4M3 = 1.0f / kFinfoAmaxE4M3;
#endif

__forceinline__ __device__ float fast_pow2(int x) {
    // We can ensure `-126 <= x and x <= 127`
    uint32_t bits_x = (x + 127) << 23;
    return *reinterpret_cast<float*>(&bits_x);
}

__forceinline__ __device__ int fast_log2_ceil(float x) {
    auto bits_x = *reinterpret_cast<uint32_t*>(&x);
    auto exp_x = (bits_x >> 23) & 0xff;
    auto man_bits = bits_x & ((1 << 23) - 1);
    return exp_x - 127 + (man_bits != 0);
}

lishen's avatar
lishen committed
368
369
370
template <bool kRoundScale>
__forceinline__ __device__ void calculate_fp8_scales(float amax, float& scale, float& scale_inv) {
    if constexpr(kRoundScale) {
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
        auto exp_scale_inv = fast_log2_ceil(amax * kFinfoAmaxInvE4M3);
        scale = fast_pow2(-exp_scale_inv);
        scale_inv = fast_pow2(exp_scale_inv);
    } else {
        scale_inv = amax * kFinfoAmaxInvE4M3;
        scale = kFinfoAmaxE4M3 / amax;
    }
}

template <bool kIsUE8M0, typename out_dtype_t = std::conditional_t<kIsUE8M0, uint8_t, float>>
__forceinline__ __device__ out_dtype_t extract_required_scale_format(float value) {
    if constexpr (kIsUE8M0) {
        return static_cast<uint8_t>((*reinterpret_cast<uint32_t*>(&value)) >> 23);
    } else {
        return value;
    }
Shifang Xu's avatar
Shifang Xu committed
387
388
}

lijian6's avatar
lijian6 committed
389
390
391
__forceinline__ __device__ int get_lane_id() {
    int lane_id = threadIdx.x % kWarpSize;
    return lane_id;
Shifang Xu's avatar
Shifang Xu committed
392
393
}

394
template <int kNumRanks, bool kSyncOnly = false>
lijian6's avatar
lijian6 committed
395
__forceinline__ __device__ void barrier_block(int **barrier_signal_ptrs, int rank) {
Chenggang Zhao's avatar
Chenggang Zhao committed
396
397
    auto thread_id = static_cast<int>(threadIdx.x);

lijian6's avatar
lijian6 committed
398
399
    // For non-sync-only cases, the memory operations by other threads in the block must be visible
    // to the `sys` scope
400
401
402
403
404
    if constexpr (not kSyncOnly) {
        memory_fence();
        __syncthreads();
    }

405
    // Add self-ranks, sub other ranks
Chenggang Zhao's avatar
Chenggang Zhao committed
406
    if (thread_id < kNumRanks) {
407
408
409
410
411
412
413
414
        atomicAdd_system(barrier_signal_ptrs[rank] + thread_id, FINISHED_SUM_TAG);
        atomicSub_system(barrier_signal_ptrs[thread_id] + rank, FINISHED_SUM_TAG);
    }
    EP_DEVICE_ASSERT(kNumRanks <= blockDim.x);

    // Check timeout
    auto start_time = clock64();
    while (true) {
lijian6's avatar
lijian6 committed
415
416
417
        auto value =
            thread_id < kNumRanks ? ld_volatile_global(barrier_signal_ptrs[rank] + thread_id) : 0;
        if (__all_sync(kFullWarpMask, value <= 0))
418
419
            break;

Chenggang Zhao's avatar
Chenggang Zhao committed
420
        if (clock64() - start_time > NUM_TIMEOUT_CYCLES and thread_id < kNumRanks) {
lijian6's avatar
lijian6 committed
421
422
            printf("DeepEP timeout check failed: rank = %d, thread = %d, value = %d)\n", rank,
                   thread_id, value);
423
424
            trap();
        }
Chenggang Zhao's avatar
Chenggang Zhao committed
425
    }
426
    __syncthreads();
Chenggang Zhao's avatar
Chenggang Zhao committed
427
}
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518

// Operation functors
template <typename T>
struct ReduceSum {
    __device__ T operator()(T a, T b) const { return a + b; }
};
template <typename T>
struct ReduceMax {
    __device__ T operator()(T a, T b) const { return a > b ? a : b; }
};
template <typename T>
struct ReduceMin {
    __device__ T operator()(T a, T b) const { return a < b ? a : b; }
};
template <typename T>
struct ReduceAnd {
    __device__ T operator()(T a, T b) const { return a & b; }
};
template <typename T>
struct ReduceOr {
    __device__ T operator()(T a, T b) const { return a | b; }
};

// Unified reduction function
template <int kNumLanesPerGroup, bool kIntergroupReduce, typename T, typename Op>
__forceinline__ __device__ T warp_reduce(T value, Op op) {
    EP_STATIC_ASSERT(kNumLanesPerGroup == kWarpSize or kNumLanesPerGroup == 32 or
                     kNumLanesPerGroup == 16 or kNumLanesPerGroup == 8 or kNumLanesPerGroup == 4 or
                     kNumLanesPerGroup == 2 or kNumLanesPerGroup == 1,
                     "Invalid number of lanes");
    constexpr uint32_t mask = 0xffffffff;
    if constexpr (kIntergroupReduce) {
        if constexpr (kNumLanesPerGroup <= 1)
        value = op(value, shfl_xor(value, 1));
        if constexpr (kNumLanesPerGroup <= 2)
        value = op(value, shfl_xor(value, 2));
        if constexpr (kNumLanesPerGroup <= 4)
        value = op(value, shfl_xor(value, 4));
        if constexpr (kNumLanesPerGroup <= 8)
        value = op(value, shfl_xor(value, 8));
        if constexpr (kNumLanesPerGroup <= 16)
        value = op(value, shfl_xor(value, 16));
        if constexpr(kWarpSize == 64){
            if constexpr (kNumLanesPerGroup <= 32)
            value = op(value, shfl_xor(value, 32));
        }
    } else {
        if constexpr(kWarpSize == 64){
            if constexpr (kNumLanesPerGroup >= kWarpSize)
            value = op(value, shfl_xor(value, 32));
        }
        if constexpr (kNumLanesPerGroup >= 32)
        value = op(value, shfl_xor(value, 16));
        if constexpr (kNumLanesPerGroup >= 16)
        value = op(value, shfl_xor(value, 8));
        if constexpr (kNumLanesPerGroup >= 8)
        value = op(value, shfl_xor(value, 4));
        if constexpr (kNumLanesPerGroup >= 4)
        value = op(value, shfl_xor(value, 2));
        if constexpr (kNumLanesPerGroup >= 2)
        value = op(value, shfl_xor(value, 1));
    }
    return value;
}

// Convenience aliases
template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_sum(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceSum<T>{});
}

template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_max(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceMax<T>{});
}

template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_min(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceMin<T>{});
}

template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_and(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceAnd<T>{});
}

template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_or(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceOr<T>{});
}

Chenggang Zhao's avatar
Chenggang Zhao committed
519
} // namespace deep_ep