utils.cuh 24.1 KB
Newer Older
lijian6's avatar
lijian6 committed
1
#include "hip/hip_runtime.h"
Chenggang Zhao's avatar
Chenggang Zhao committed
2
#pragma once
lijian6's avatar
lijian6 committed
3
#include "configs.cuh"
Chenggang Zhao's avatar
Chenggang Zhao committed
4
5
#include "exception.cuh"

lijian6's avatar
lijian6 committed
6
7
8
9
10
11
12
13
14
15
16
17
18
#define UNROLLED_WARP_COPY(UNROLL_FACTOR, LANE_ID, N, DST, SRC, LD_FUNC, ST_FUNC)                  \
    {                                                                                              \
        constexpr int kLoopStride = kWarpSize * (UNROLL_FACTOR);                                   \
        typename std::remove_reference<decltype(LD_FUNC((SRC) + 0))>::type                         \
             unrolled_values[(UNROLL_FACTOR)];                                                     \
        auto __src = (SRC);                                                                        \
        auto __dst = (DST);                                                                        \
        for (int __i = (LANE_ID); __i < ((N) / kLoopStride) * kLoopStride; __i += kLoopStride) {   \
            _Pragma("unroll") for (int __j = 0; __j < (UNROLL_FACTOR); ++__j)                      \
                unrolled_values[__j] = LD_FUNC(__src + __i + __j * kWarpSize);                     \
            _Pragma("unroll") for (int __j = 0; __j < (UNROLL_FACTOR); ++__j)                      \
                ST_FUNC(__dst + __i + __j * kWarpSize, unrolled_values[__j]);                      \
        }                                                                                          \
lijian6's avatar
lijian6 committed
19
20
        for (int __i = ((N) / kLoopStride) * kLoopStride + (LANE_ID); __i < (N); __i += kWarpSize) \
            ST_FUNC(__dst + __i, LD_FUNC(__src + __i)); \
lijian6's avatar
lijian6 committed
21
22
    }

lishen's avatar
lishen committed
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
#define UNROLLED_WARP_COPY_LL(UNROLL_FACTOR, LANE_ID, N, DST, SRC, LD_FUNC, ST_FUNC)                                                        \
    {                                                                                                                                       \
        constexpr int kLoopStride = kWarpSize * (UNROLL_FACTOR);                                                                            \
        typename std::remove_reference<decltype(LD_FUNC((SRC) + 0))>::type unrolled_values[(UNROLL_FACTOR)];                                \
        auto __src = (SRC);                                                                                                                 \
        auto __dst = (DST);                                                                                                                 \
        for(int __i = (LANE_ID); __i < ((N) / kLoopStride) * kLoopStride; __i += kLoopStride) {                                             \
            _Pragma("unroll") for(int __j = 0; __j < (UNROLL_FACTOR); ++__j) unrolled_values[__j] = LD_FUNC(__src + __i + __j * kWarpSize); \
            _Pragma("unroll") for(int __j = 0; __j < (UNROLL_FACTOR); ++__j) ST_FUNC(__dst + __i + __j * kWarpSize, unrolled_values[__j]);  \
        }                                                                                                                                   \
        for(int __i = ((N) / kLoopStride) * kLoopStride + (LANE_ID); __i < (N); __i += kWarpSize)                                           \
            ST_FUNC(__dst + __i, LD_FUNC(__src + __i));                                                                                     \
    }


lijian6's avatar
lijian6 committed
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
#define UNROLLED_WARP_COPY_EMULATED(UNROLL_FACTOR, LANE_ID, N, DST, SRC, LD_FUNC, ST_FUNC)         \
    {                                                                                              \
        constexpr int kLoopStride = kEmulatedWarpSize * (UNROLL_FACTOR);                           \
        typename std::remove_reference<decltype(LD_FUNC((SRC) + 0))>::type                         \
             unrolled_values[(UNROLL_FACTOR)];                                                     \
        auto __src = (SRC);                                                                        \
        auto __dst = (DST);                                                                        \
        for (int __i = (LANE_ID); __i < ((N) / kLoopStride) * kLoopStride; __i += kLoopStride) {   \
            _Pragma("unroll") for (int __j = 0; __j < (UNROLL_FACTOR); ++__j)                      \
                unrolled_values[__j] = LD_FUNC(__src + __i + __j * kEmulatedWarpSize);             \
            _Pragma("unroll") for (int __j = 0; __j < (UNROLL_FACTOR); ++__j)                      \
                ST_FUNC(__dst + __i + __j * kEmulatedWarpSize, unrolled_values[__j]);              \
        }                                                                                          \
        for (int __i = ((N) / kLoopStride) * kLoopStride + (LANE_ID); __i < (N);                   \
             __i += kEmulatedWarpSize)                                                             \
            ST_FUNC(__dst + __i, LD_FUNC(__src + __i));                                            \
    }
// HELPER FUNCTIONS
// #####################################################################################
lishen's avatar
lishen committed
57
#define DEVICE_INLINE __device__ inline __attribute__((always_inline))
lijian6's avatar
lijian6 committed
58
59
60
61
62

template <typename T>
__device__ __forceinline__ T shfl_xor(const T val, int laneMask, int width = kWarpSize,
                                      uint64_t shfl_sync_mask = kFullWarpMask) {
    return __shfl_xor(val, laneMask, width);
Chenggang Zhao's avatar
Chenggang Zhao committed
63
64
}

65
66
67
template <typename T>
__device__ __forceinline__ T shfl_sync(const T val, int srcLane = 0, int width = kWarpSize,
                                       uint64_t shfl_sync_mask = kFullWarpMask) { // Let compiler deduce type
lijian6's avatar
lijian6 committed
68
69
70
71
72
73
74
    return __shfl(val, srcLane, width);
}

__device__ __forceinline__ int __any_sync(uint64_t mask, int predicate) {
    uint64_t predicate_bit_pattern = __ballot(predicate);
    return (predicate_bit_pattern & mask) > 0;
}
Chenggang Zhao's avatar
Chenggang Zhao committed
75

lijian6's avatar
lijian6 committed
76
77
78
79
__device__ __forceinline__ int __all_sync(uint64_t mask, int predicate) {
    uint64_t predicate_bit_pattern = __ballot(predicate);
    return (~predicate_bit_pattern & mask) == 0;
}
Chenggang Zhao's avatar
Chenggang Zhao committed
80

lijian6's avatar
lijian6 committed
81
82
83
84
85
86
__device__ __forceinline__ void syncwarp() {
    __builtin_amdgcn_fence(__ATOMIC_RELEASE, "wavefront");
    __builtin_amdgcn_wave_barrier();
    __builtin_amdgcn_fence(__ATOMIC_ACQUIRE, "wavefront");
}
// ######################################################################################################
87

lijian6's avatar
lijian6 committed
88
namespace deep_ep {
89

lijian6's avatar
lijian6 committed
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
template <int kBytes> struct VecInt {};
template <> struct VecInt<1> {
    using vec_t = int8_t;
};
template <> struct VecInt<2> {
    using vec_t = int16_t;
};
template <> struct VecInt<4> {
    using vec_t = int;
};
template <> struct VecInt<8> {
    using vec_t = int64_t;
};
template <> struct VecInt<16> {
    using native_int4 = int __attribute__((ext_vector_type(4)));
    using vec_t       = native_int4;
106
107
};

108
109
110
111
112
113
114
115
116
template <typename FuncT>
struct PatternVisitor {
    FuncT func;

    __device__ __host__ explicit PatternVisitor(FuncT&& func) : func(std::forward<FuncT>(func)) {}

    __device__ __host__ auto operator[](const uint32_t& i) { return func(i); }
};

Chenggang Zhao's avatar
Chenggang Zhao committed
117
__device__ __forceinline__ void trap() {
lijian6's avatar
lijian6 committed
118
    abort();
Chenggang Zhao's avatar
Chenggang Zhao committed
119
120
121
}

__device__ __forceinline__ void memory_fence() {
lijian6's avatar
lijian6 committed
122
    __threadfence_system();
Chenggang Zhao's avatar
Chenggang Zhao committed
123
124
125
}

__device__ __forceinline__ void memory_fence_gpu() {
lijian6's avatar
lijian6 committed
126
    __threadfence();
Chenggang Zhao's avatar
Chenggang Zhao committed
127
128
129
}

__device__ __forceinline__ void memory_fence_cta() {
lijian6's avatar
lijian6 committed
130
    __threadfence_block();
Chenggang Zhao's avatar
Chenggang Zhao committed
131
132
}

lijian6's avatar
lijian6 committed
133
__device__ __forceinline__ void st_relaxed_sys_global(int *ptr, int val) {
lijian6's avatar
lijian6 committed
134
    __hip_atomic_store(ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
135
136
}

lijian6's avatar
lijian6 committed
137
138
__device__ __forceinline__ void st_release_sys_global(const int *ptr, int val) {
    __hip_atomic_store(const_cast<int *>(ptr), val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
139
140
}

141
142
143
144
__device__ __forceinline__ void st_release_sys_global(const int64_t *ptr, int64_t val) {
    __hip_atomic_store(const_cast<int64_t *>(ptr), val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_SYSTEM);
}

lijian6's avatar
lijian6 committed
145
146
147
148
149
__device__ __forceinline__ void st_release_cta(const int *ptr, int val) {
    __hip_atomic_store(const_cast<int *>(ptr), val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_WORKGROUP);
}

__device__ __forceinline__ int ld_relaxed_sys_global(const int *ptr) {
lijian6's avatar
lijian6 committed
150
151
152
    int ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
    return ret;
lijian6's avatar
lijian6 committed
153
}
lishen's avatar
lishen committed
154

lijian6's avatar
lijian6 committed
155
156
157
158
__device__ __forceinline__ int ld_relaxed_sys_global(const uint64_t *ptr) {
    uint64_t ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
    return ret;
Chenggang Zhao's avatar
Chenggang Zhao committed
159
}
lishen's avatar
lishen committed
160

lijian6's avatar
lijian6 committed
161
162
163
164
165
__device__ __forceinline__ int ld_relaxed_sys_global(const int64_t *ptr) {
    int64_t ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
    return ret;
}
Chenggang Zhao's avatar
Chenggang Zhao committed
166
167
168

__device__ __forceinline__ int ld_acquire_sys_global(const int *ptr) {
    int ret;
lijian6's avatar
lijian6 committed
169
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
170
171
172
173
174
    return ret;
}

__device__ __forceinline__ uint64_t ld_acquire_sys_global(const uint64_t *ptr) {
    uint64_t ret;
lijian6's avatar
lijian6 committed
175
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
176
177
178
    return ret;
}

lijian6's avatar
lijian6 committed
179
180
181
182
183
184
__device__ __forceinline__ int64_t ld_acquire_sys_global(const int64_t *ptr) {
    int64_t ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_SYSTEM);
    return ret;
}

Chenggang Zhao's avatar
Chenggang Zhao committed
185
186
__device__ __forceinline__ int ld_acquire_global(const int *ptr) {
    int ret;
lijian6's avatar
lijian6 committed
187
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
188
189
190
    return ret;
}

191
192
193
194
195
196
__device__ __forceinline__ int64_t ld_acquire_global(const int64_t *ptr) {
    int64_t ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_AGENT);
    return ret;
}

lijian6's avatar
lijian6 committed
197
__device__ __forceinline__ int atomic_add_release_global(const int *ptr, int value) {
Chenggang Zhao's avatar
Chenggang Zhao committed
198
    int ret;
lishen's avatar
lishen committed
199
200
    ret = __hip_atomic_fetch_add(const_cast<int *>(ptr), value, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
    // ret = atomicAdd((int*)ptr, value);
Chenggang Zhao's avatar
Chenggang Zhao committed
201
202
203
    return ret;
}

204
205
206
207
208
209
__device__ __forceinline__ int ld_relaxed_global(const int *ptr) {
    int ret;
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
    return ret;
}

Chenggang Zhao's avatar
Chenggang Zhao committed
210
211
__device__ __forceinline__ int ld_acquire_cta(const int *ptr) {
    int ret;
lijian6's avatar
lijian6 committed
212
    ret = __hip_atomic_load(ptr, __ATOMIC_ACQUIRE, __HIP_MEMORY_SCOPE_WORKGROUP);
Chenggang Zhao's avatar
Chenggang Zhao committed
213
214
215
    return ret;
}

lijian6's avatar
lijian6 committed
216
__device__ __forceinline__ int ld_volatile_global(const volatile int *ptr) {
Chenggang Zhao's avatar
Chenggang Zhao committed
217
    int ret;
lijian6's avatar
lijian6 committed
218
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
219
220
221
    return ret;
}

lijian6's avatar
lijian6 committed
222
__device__ __forceinline__ float ld_volatile_global(const volatile float *ptr) {
Chenggang Zhao's avatar
Chenggang Zhao committed
223
    float ret;
lijian6's avatar
lijian6 committed
224
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
225
226
227
    return ret;
}

lijian6's avatar
lijian6 committed
228
__device__ __forceinline__ int64_t ld_volatile_global(const volatile int64_t *ptr) {
Chenggang Zhao's avatar
Chenggang Zhao committed
229
    int64_t ret;
lijian6's avatar
lijian6 committed
230
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
231
232
233
    return ret;
}

lijian6's avatar
lijian6 committed
234
__device__ __forceinline__ int64_t ld_volatile_global(const volatile uint64_t *ptr) {
Chenggang Zhao's avatar
Chenggang Zhao committed
235
    int64_t ret;
lijian6's avatar
lijian6 committed
236
    ret = __hip_atomic_load(ptr, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_SYSTEM);
Chenggang Zhao's avatar
Chenggang Zhao committed
237
238
239
    return ret;
}

lijian6's avatar
lijian6 committed
240
template <typename dtype_t> __device__ __forceinline__ dtype_t ld_nc_global(const dtype_t *ptr) {
lijian6's avatar
lijian6 committed
241
242
243
    using T  = typename VecInt<sizeof(dtype_t)>::vec_t;
    auto ret = __builtin_nontemporal_load(reinterpret_cast<const T *>(ptr));
    return *reinterpret_cast<dtype_t *>(&ret);
Chenggang Zhao's avatar
Chenggang Zhao committed
244
245
}

lishen's avatar
lishen committed
246
247
248
249
250
251
template <typename dtype_t> __device__ __forceinline__ dtype_t ld_direct_global(const dtype_t *ptr) {
    using T  = typename VecInt<sizeof(dtype_t)>::vec_t;
    auto ret = *(reinterpret_cast<const T *>(ptr));
    return *reinterpret_cast<dtype_t *>(&ret);
}

lijian6's avatar
lijian6 committed
252
////////////////// used in ibgda
Chenggang Zhao's avatar
Chenggang Zhao committed
253
__device__ __forceinline__ void st_na_relaxed(const uint8_t *ptr, uint8_t val) {
lijian6's avatar
lijian6 committed
254
255
    uint8_t *non_const_ptr = const_cast<uint8_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
256
257
258
}

__device__ __forceinline__ void st_na_relaxed(const uint16_t *ptr, uint16_t val) {
lijian6's avatar
lijian6 committed
259
260
    uint16_t *non_const_ptr = const_cast<uint16_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
261
262
263
}

__device__ __forceinline__ void st_na_relaxed(const uint32_t *ptr, uint32_t val) {
lijian6's avatar
lijian6 committed
264
265
    uint32_t *non_const_ptr = const_cast<uint32_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
266
267
268
}

__device__ __forceinline__ void st_na_relaxed(const int *ptr, int val) {
lijian6's avatar
lijian6 committed
269
270
    int *non_const_ptr = const_cast<int *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
271
272
}

lishen's avatar
lishen committed
273
274
275
276
277
278
279
280
281
282
__device__ __forceinline__ void st_na_relaxed(const float *ptr, float val) {
    float *non_const_ptr = const_cast<float *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
}

__device__ __forceinline__ void st_na_relaxed(const int64_t *ptr, int64_t val) {
    int64_t *non_const_ptr = const_cast<int64_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
}

Chenggang Zhao's avatar
Chenggang Zhao committed
283
__device__ __forceinline__ void st_na_relaxed(const int4 *ptr, int4 val) {
lijian6's avatar
lijian6 committed
284
    int4 *non_const_ptr = const_cast<int4 *>(ptr);
lishen's avatar
lishen committed
285
286
287
288
    __hip_atomic_store(&(non_const_ptr->x), val.x, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
    __hip_atomic_store(&(non_const_ptr->y), val.y, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
    __hip_atomic_store(&(non_const_ptr->z), val.z, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
    __hip_atomic_store(&(non_const_ptr->w), val.w, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
289
290
291
}

__device__ __forceinline__ void st_na_release(const int *ptr, int val) {
lijian6's avatar
lijian6 committed
292
293
    int *non_const_ptr = const_cast<int *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELAXED, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
294
295
296
}

__device__ __forceinline__ void st_na_release(const uint32_t *ptr, uint32_t val) {
lijian6's avatar
lijian6 committed
297
298
    uint32_t *non_const_ptr = const_cast<uint32_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
299
300
301
}

__device__ __forceinline__ void st_na_release(const uint64_t *ptr, uint64_t val) {
lijian6's avatar
lijian6 committed
302
303
    uint64_t *non_const_ptr = const_cast<uint64_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
Chenggang Zhao's avatar
Chenggang Zhao committed
304
305
}

306
307
308
309
310
__device__ __forceinline__ void st_na_release(const int64_t *ptr, int64_t val) {
    int64_t *non_const_ptr = const_cast<int64_t *>(ptr);
    __hip_atomic_store(non_const_ptr, val, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
}

lishen's avatar
lishen committed
311
312
313
314
315
316
317
318
__device__ __forceinline__ void st_na_release(const int4 *ptr, int4 val) {
    int4 *non_const_ptr = const_cast<int4 *>(ptr);
    __hip_atomic_store(&(non_const_ptr->x), val.x, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
    __hip_atomic_store(&(non_const_ptr->y), val.y, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
    __hip_atomic_store(&(non_const_ptr->z), val.z, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
    __hip_atomic_store(&(non_const_ptr->w), val.w, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_AGENT);
}

lijian6's avatar
lijian6 committed
319
// TODO:: apply "st.global.L1::no_allocate" in ROCM
Chenggang Zhao's avatar
Chenggang Zhao committed
320
template <typename dtype_t>
lijian6's avatar
lijian6 committed
321
322
323
__device__ __forceinline__ void st_na_global(const dtype_t *ptr, const dtype_t &value) {
    st_na_global(reinterpret_cast<const typename VecInt<sizeof(dtype_t)>::vec_t *>(ptr),
                 *reinterpret_cast<const typename VecInt<sizeof(dtype_t)>::vec_t *>(&value));
Chenggang Zhao's avatar
Chenggang Zhao committed
324
325
}

lijian6's avatar
lijian6 committed
326
327
328
template <> __device__ __forceinline__ void st_na_global(const int *ptr, const int &value) {
    int *non_const_ptr = const_cast<int *>(ptr);
    *non_const_ptr     = value;
329
330
}

lijian6's avatar
lijian6 committed
331
332
333
template <> __device__ __forceinline__ void st_na_global(const int64_t *ptr, const int64_t &value) {
    int64_t *non_const_ptr = const_cast<int64_t *>(ptr);
    *non_const_ptr         = value;
Chenggang Zhao's avatar
Chenggang Zhao committed
334
335
}

lijian6's avatar
lijian6 committed
336
337
338
template <> __device__ __forceinline__ void st_na_global(const float *ptr, const float &value) {
    float *non_const_ptr = const_cast<float *>(ptr);
    *non_const_ptr       = value;
Chenggang Zhao's avatar
Chenggang Zhao committed
339
340
}

lijian6's avatar
lijian6 committed
341
342
343
template <> __device__ __forceinline__ void st_na_global(const int4 *ptr, const int4 &value) {
    int4 *non_const_ptr = const_cast<int4 *>(ptr);
    *non_const_ptr      = value;
344
345
}

Chenggang Zhao's avatar
Chenggang Zhao committed
346
__forceinline__ __device__ void get_channel_task_range(int num_tokens, int num_sms, int sm_id,
lijian6's avatar
lijian6 committed
347
348
349
350
                                                       int &token_start_idx, int &token_end_idx) {
    int num_tokens_per_sm = DIVUP(num_tokens, num_sms);
    token_start_idx       = min(num_tokens_per_sm * sm_id, num_tokens);
    token_end_idx         = min(token_start_idx + num_tokens_per_sm, num_tokens);
Chenggang Zhao's avatar
Chenggang Zhao committed
351
352
}

353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
template <typename dtype_a_t, typename dtype_b_t>
__device__ __forceinline__ dtype_b_t pack2(const dtype_a_t& x, const dtype_a_t& y) {
    EP_STATIC_ASSERT(sizeof(dtype_a_t) * 2 == sizeof(dtype_b_t), "Invalid dtypes");
    dtype_b_t packed;
    auto unpacked_ptr = reinterpret_cast<dtype_a_t*>(&packed);
    unpacked_ptr[0] = x, unpacked_ptr[1] = y;
    return packed;
}

template <typename dtype_a_t, typename dtype_b_t>
__device__ __forceinline__ void unpack2(const dtype_b_t& packed, dtype_a_t& x, dtype_a_t& y) {
    EP_STATIC_ASSERT(sizeof(dtype_a_t) * 2 == sizeof(dtype_b_t), "Invalid dtypes");
    auto unpacked_ptr = reinterpret_cast<const dtype_a_t*>(&packed);
    x = unpacked_ptr[0], y = unpacked_ptr[1];
}

Chenggang Zhao's avatar
Chenggang Zhao committed
369
template <typename dtype_t>
lijian6's avatar
lijian6 committed
370
__device__ __forceinline__ dtype_t broadcast(dtype_t &ptr, int src_lane_idx) {
Chenggang Zhao's avatar
Chenggang Zhao committed
371
    EP_STATIC_ASSERT(sizeof(dtype_t) % sizeof(int) == 0, "");
lijian6's avatar
lijian6 committed
372
373
374
375
376
377
378
379
    auto send_int_values = reinterpret_cast<int *>(&ptr);
    int  recv_int_values[sizeof(dtype_t) / sizeof(int)];
#pragma unroll
    for (int i = 0; i < sizeof(dtype_t) / sizeof(int); ++i)
        recv_int_values[i] = shfl_sync(send_int_values[i], src_lane_idx);
    return *reinterpret_cast<dtype_t *>(recv_int_values);
}

380
// 设置不同的量化方式的最大值与相反数
lishen's avatar
lishen committed
381
constexpr float kFinfoAmaxE4M3 = 448.0f;
lishen's avatar
lishen committed
382
constexpr float kFinfoAmaxInvE4M3 = 1.0f / kFinfoAmaxE4M3;
383
384
385
386
constexpr float kFinfoAmaxE5M2 = 57344.0f; 
constexpr float kFinfoAmaxInvE5M2 = 1.0f / kFinfoAmaxE5M2;
constexpr float kFinfoAmaxInt8 = 127.0f;
constexpr float kFinfoAmaxInvInt8 = 1.0f / 127.0f;
387
388
389
390
391
392
393
394
395
396
397
398
399
400

__forceinline__ __device__ float fast_pow2(int x) {
    // We can ensure `-126 <= x and x <= 127`
    uint32_t bits_x = (x + 127) << 23;
    return *reinterpret_cast<float*>(&bits_x);
}

__forceinline__ __device__ int fast_log2_ceil(float x) {
    auto bits_x = *reinterpret_cast<uint32_t*>(&x);
    auto exp_x = (bits_x >> 23) & 0xff;
    auto man_bits = bits_x & ((1 << 23) - 1);
    return exp_x - 127 + (man_bits != 0);
}

401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
template <int kQuantType>
__forceinline__ __device__ void calculate_quant8bit_scales(float amax, float& scale, float& scale_inv, bool round_scale=0) {
    amax = fmaxf(amax, 1e-6f);
    if constexpr(kQuantType == 1) { // 使用 INT8 对称量化
        scale_inv = kFinfoAmaxInvInt8 * amax;
        scale = kFinfoAmaxInt8 / amax;
    } else if constexpr(kQuantType == 2 || kQuantType == 3) {   // 使用 FP8_E4M3 或 FP8_UE8M0 非对称量化
        if (round_scale) {
            auto exp_scale_inv = fast_log2_ceil(amax * kFinfoAmaxInvE4M3);
            scale = fast_pow2(-exp_scale_inv);
            scale_inv = fast_pow2(exp_scale_inv);
        } else {
            scale_inv = amax * kFinfoAmaxInvE4M3;
            scale = kFinfoAmaxE4M3 / amax;
        }
    } else if constexpr(kQuantType == 4) { // 使用 FP8_E5M2 对称量化
        if (round_scale) {
            auto exp_scale_inv = fast_log2_ceil(amax * kFinfoAmaxInvE5M2);
            scale = fast_pow2(-exp_scale_inv);
            scale_inv = fast_pow2(exp_scale_inv);
        } else {
            scale_inv = amax * kFinfoAmaxInvE5M2;
            scale = kFinfoAmaxE5M2 / amax;
        }
425
426
427
428
429
430
431
432
433
434
    }
}

template <bool kIsUE8M0, typename out_dtype_t = std::conditional_t<kIsUE8M0, uint8_t, float>>
__forceinline__ __device__ out_dtype_t extract_required_scale_format(float value) {
    if constexpr (kIsUE8M0) {
        return static_cast<uint8_t>((*reinterpret_cast<uint32_t*>(&value)) >> 23);
    } else {
        return value;
    }
Shifang Xu's avatar
Shifang Xu committed
435
436
}

lijian6's avatar
lijian6 committed
437
438
439
__forceinline__ __device__ int get_lane_id() {
    int lane_id = threadIdx.x % kWarpSize;
    return lane_id;
Shifang Xu's avatar
Shifang Xu committed
440
441
}

442
template <int kNumRanks, bool kSyncOnly = false>
lijian6's avatar
lijian6 committed
443
__forceinline__ __device__ void barrier_block(int **barrier_signal_ptrs, int rank) {
Chenggang Zhao's avatar
Chenggang Zhao committed
444
445
    auto thread_id = static_cast<int>(threadIdx.x);

lijian6's avatar
lijian6 committed
446
447
    // For non-sync-only cases, the memory operations by other threads in the block must be visible
    // to the `sys` scope
448
449
450
451
452
    if constexpr (not kSyncOnly) {
        memory_fence();
        __syncthreads();
    }

453
    // Add self-ranks, sub other ranks
Chenggang Zhao's avatar
Chenggang Zhao committed
454
    if (thread_id < kNumRanks) {
455
456
457
458
459
460
461
462
        atomicAdd_system(barrier_signal_ptrs[rank] + thread_id, FINISHED_SUM_TAG);
        atomicSub_system(barrier_signal_ptrs[thread_id] + rank, FINISHED_SUM_TAG);
    }
    EP_DEVICE_ASSERT(kNumRanks <= blockDim.x);

    // Check timeout
    auto start_time = clock64();
    while (true) {
lijian6's avatar
lijian6 committed
463
464
465
        auto value =
            thread_id < kNumRanks ? ld_volatile_global(barrier_signal_ptrs[rank] + thread_id) : 0;
        if (__all_sync(kFullWarpMask, value <= 0))
466
467
            break;

Chenggang Zhao's avatar
Chenggang Zhao committed
468
        if (clock64() - start_time > NUM_TIMEOUT_CYCLES and thread_id < kNumRanks) {
lijian6's avatar
lijian6 committed
469
470
            printf("DeepEP timeout check failed: rank = %d, thread = %d, value = %d)\n", rank,
                   thread_id, value);
471
472
            trap();
        }
Chenggang Zhao's avatar
Chenggang Zhao committed
473
    }
474
    __syncthreads();
Chenggang Zhao's avatar
Chenggang Zhao committed
475
}
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566

// Operation functors
template <typename T>
struct ReduceSum {
    __device__ T operator()(T a, T b) const { return a + b; }
};
template <typename T>
struct ReduceMax {
    __device__ T operator()(T a, T b) const { return a > b ? a : b; }
};
template <typename T>
struct ReduceMin {
    __device__ T operator()(T a, T b) const { return a < b ? a : b; }
};
template <typename T>
struct ReduceAnd {
    __device__ T operator()(T a, T b) const { return a & b; }
};
template <typename T>
struct ReduceOr {
    __device__ T operator()(T a, T b) const { return a | b; }
};

// Unified reduction function
template <int kNumLanesPerGroup, bool kIntergroupReduce, typename T, typename Op>
__forceinline__ __device__ T warp_reduce(T value, Op op) {
    EP_STATIC_ASSERT(kNumLanesPerGroup == kWarpSize or kNumLanesPerGroup == 32 or
                     kNumLanesPerGroup == 16 or kNumLanesPerGroup == 8 or kNumLanesPerGroup == 4 or
                     kNumLanesPerGroup == 2 or kNumLanesPerGroup == 1,
                     "Invalid number of lanes");
    constexpr uint32_t mask = 0xffffffff;
    if constexpr (kIntergroupReduce) {
        if constexpr (kNumLanesPerGroup <= 1)
        value = op(value, shfl_xor(value, 1));
        if constexpr (kNumLanesPerGroup <= 2)
        value = op(value, shfl_xor(value, 2));
        if constexpr (kNumLanesPerGroup <= 4)
        value = op(value, shfl_xor(value, 4));
        if constexpr (kNumLanesPerGroup <= 8)
        value = op(value, shfl_xor(value, 8));
        if constexpr (kNumLanesPerGroup <= 16)
        value = op(value, shfl_xor(value, 16));
        if constexpr(kWarpSize == 64){
            if constexpr (kNumLanesPerGroup <= 32)
            value = op(value, shfl_xor(value, 32));
        }
    } else {
        if constexpr(kWarpSize == 64){
            if constexpr (kNumLanesPerGroup >= kWarpSize)
            value = op(value, shfl_xor(value, 32));
        }
        if constexpr (kNumLanesPerGroup >= 32)
        value = op(value, shfl_xor(value, 16));
        if constexpr (kNumLanesPerGroup >= 16)
        value = op(value, shfl_xor(value, 8));
        if constexpr (kNumLanesPerGroup >= 8)
        value = op(value, shfl_xor(value, 4));
        if constexpr (kNumLanesPerGroup >= 4)
        value = op(value, shfl_xor(value, 2));
        if constexpr (kNumLanesPerGroup >= 2)
        value = op(value, shfl_xor(value, 1));
    }
    return value;
}

// Convenience aliases
template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_sum(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceSum<T>{});
}

template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_max(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceMax<T>{});
}

template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_min(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceMin<T>{});
}

template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_and(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceAnd<T>{});
}

template <int kNumLanesPerGroup = kWarpSize, bool kIntergroupReduce = false, typename T>
__forceinline__ __device__ T warp_reduce_or(T value) {
    return warp_reduce<kNumLanesPerGroup, kIntergroupReduce, T>(value, ReduceOr<T>{});
}

Chenggang Zhao's avatar
Chenggang Zhao committed
567
} // namespace deep_ep