Merge pull request #203 from ROCm/merge_from_public

Merge from public

Merge pull request #203 from ROCm/merge_from_public
Merge from public
f221c2b0 · Illia Silin · GitHub · 140d2fa6 · e1cd4121 · f221c2b0
Unverified Commit f221c2b0 authored Oct 24, 2024 by Illia Silin Committed by GitHub Oct 24, 2024
14 changed files
--- a/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/dl/device_grouped_conv3d_bwd_weight_dl_gndhwc_gkzyxc_gndhwk_bf16_instance.cpp
+++ b/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/dl/device_grouped_conv3d_bwd_weight_dl_gndhwc_gkzyxc_gndhwk_bf16_instance.cpp
 // SPDX-License-Identifier: MIT
-// Copyright (c) 2018-2023, Advanced Micro Devices, Inc. All rights reserved.
+// Copyright (c) 2018-2024, Advanced Micro Devices, Inc. All rights reserved.
 #include "ck/library/tensor_operation_instance/add_device_operation_instance.hpp"
 #include "ck/library/tensor_operation_instance/gpu/grouped_conv_bwd_weight/device_grouped_conv_bwd_weight_dl_instance.hpp"

--- a/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/dl/device_grouped_conv3d_bwd_weight_dl_ndhwgc_gkzyxc_ndhwgk_bf16_instance.cpp
+++ b/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/dl/device_grouped_conv3d_bwd_weight_dl_ndhwgc_gkzyxc_ndhwgk_bf16_instance.cpp
 // SPDX-License-Identifier: MIT
-// Copyright (c) 2018-2023, Advanced Micro Devices, Inc. All rights reserved.
+// Copyright (c) 2018-2024, Advanced Micro Devices, Inc. All rights reserved.
 #include "ck/library/tensor_operation_instance/add_device_operation_instance.hpp"
 #include "ck/library/tensor_operation_instance/gpu/grouped_conv_bwd_weight/device_grouped_conv_bwd_weight_dl_instance.hpp"

--- a/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/xdl/device_grouped_conv3d_bwd_weight_two_stage_xdl_ndhwgc_gkzyxc_ndhwgk_bf16_pipev2_instance.cpp
+++ b/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/xdl/device_grouped_conv3d_bwd_weight_two_stage_xdl_ndhwgc_gkzyxc_ndhwgk_bf16_pipev2_instance.cpp
+// SPDX-License-Identifier: MIT
+// Copyright (c) 2024, Advanced Micro Devices, Inc. All rights reserved.
+#include "ck/library/tensor_operation_instance/add_device_operation_instance.hpp"
+#include "ck/library/tensor_operation_instance/gpu/grouped_conv_bwd_weight/device_grouped_conv_bwd_weight_two_stage_xdl_instance.hpp"
+namespace ck {
+namespace tensor_operation {
+namespace device {
+namespace instance {
+// Compilation parameters for in[n, hi, wi, g, c] * wei[g, k, y, x, c] = out[n, ho, wo, g, k]
+void add_device_grouped_conv3d_bwd_weight_two_stage_xdl_ndhwgc_gkzyxc_ndhwgk_bf16_pipev2_instances(
+    std::vector<std::unique_ptr<DeviceGroupedConvBwdWeight<3,
+                                                           NDHWGC,
+                                                           GKZYXC,
+                                                           NDHWGK,
+                                                           BF16,
+                                                           BF16,
+                                                           BF16,
+                                                           PassThrough,
+                                                           PassThrough,
+                                                           PassThrough>>>& instances)
+{
+    // 1. Default
+    add_device_operation_instances(
+        instances,
+        device_grouped_conv_bwd_weight_two_stage_xdl_c_shuffle_bf16_instances<
+            3,
+            NDHWGC,
+            GKZYXC,
+            NDHWGK,
+            ConvBwdWeightDefault,
+            BlockGemmPipelineScheduler::Intrawave,
+            BlockGemmPipelineVersion::v2>{});
+}
+} // namespace instance
+} // namespace device
+} // namespace tensor_operation
+} // namespace ck
--- a/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/xdl/device_grouped_conv3d_bwd_weight_two_stage_xdl_ndhwgc_gkzyxc_ndhwgk_bf16_pipev5_instance.cpp
+++ b/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/xdl/device_grouped_conv3d_bwd_weight_two_stage_xdl_ndhwgc_gkzyxc_ndhwgk_bf16_pipev5_instance.cpp
+// SPDX-License-Identifier: MIT
+// Copyright (c) 2024, Advanced Micro Devices, Inc. All rights reserved.
+#include "ck/library/tensor_operation_instance/add_device_operation_instance.hpp"
+#include "ck/library/tensor_operation_instance/gpu/grouped_conv_bwd_weight/device_grouped_conv_bwd_weight_two_stage_xdl_instance.hpp"
+namespace ck {
+namespace tensor_operation {
+namespace device {
+namespace instance {
+// Compilation parameters for in[n, hi, wi, g, c] * wei[g, k, y, x, c] = out[n, ho, wo, g, k]
+void add_device_grouped_conv3d_bwd_weight_two_stage_xdl_ndhwgc_gkzyxc_ndhwgk_bf16_pipev5_instances(
+    std::vector<std::unique_ptr<DeviceGroupedConvBwdWeight<3,
+                                                           NDHWGC,
+                                                           GKZYXC,
+                                                           NDHWGK,
+                                                           BF16,
+                                                           BF16,
+                                                           BF16,
+                                                           PassThrough,
+                                                           PassThrough,
+                                                           PassThrough>>>& instances)
+{
+    // 1. Default
+    add_device_operation_instances(
+        instances,
+        device_grouped_conv_bwd_weight_two_stage_xdl_c_shuffle_bf16_instances<
+            3,
+            NDHWGC,
+            GKZYXC,
+            NDHWGK,
+            ConvBwdWeightDefault,
+            BlockGemmPipelineScheduler::Intrawave,
+            BlockGemmPipelineVersion::v5>{});
+}
+} // namespace instance
+} // namespace device
+} // namespace tensor_operation
+} // namespace ck
--- a/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/xdl/device_grouped_conv3d_bwd_weight_two_stage_xdl_ngcdhw_gkzyxc_ngkdhw_bf16_pipev2_instance.cpp
+++ b/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/xdl/device_grouped_conv3d_bwd_weight_two_stage_xdl_ngcdhw_gkzyxc_ngkdhw_bf16_pipev2_instance.cpp
+// SPDX-License-Identifier: MIT
+// Copyright (c) 2024, Advanced Micro Devices, Inc. All rights reserved.
+#include "ck/library/tensor_operation_instance/add_device_operation_instance.hpp"
+#include "ck/library/tensor_operation_instance/gpu/grouped_conv_bwd_weight/device_grouped_conv_bwd_weight_two_stage_xdl_instance.hpp"
+namespace ck {
+namespace tensor_operation {
+namespace device {
+namespace instance {
+// Compilation parameters for in[n, hi, wi, g, c] * wei[g, k, y, x, c] = out[n, ho, wo, g, k]
+void add_device_grouped_conv3d_bwd_weight_two_stage_xdl_ngcdhw_gkzyxc_ngkdhw_bf16_pipev2_instances(
+    std::vector<std::unique_ptr<DeviceGroupedConvBwdWeight<3,
+                                                           NGCDHW,
+                                                           GKZYXC,
+                                                           NGKDHW,
+                                                           BF16,
+                                                           BF16,
+                                                           BF16,
+                                                           PassThrough,
+                                                           PassThrough,
+                                                           PassThrough>>>& instances)
+{
+    // 1. Default
+    add_device_operation_instances(
+        instances,
+        device_grouped_conv_bwd_weight_two_stage_ngchw_xdl_c_shuffle_bf16_instances<
+            3,
+            NGCDHW,
+            GKZYXC,
+            NGKDHW,
+            ConvBwdWeightDefault,
+            BlockGemmPipelineScheduler::Intrawave,
+            BlockGemmPipelineVersion::v2>{});
+}
+} // namespace instance
+} // namespace device
+} // namespace tensor_operation
+} // namespace ck
--- a/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/xdl/device_grouped_conv3d_bwd_weight_two_stage_xdl_ngcdhw_gkzyxc_ngkdhw_bf16_pipev5_instance.cpp
+++ b/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/xdl/device_grouped_conv3d_bwd_weight_two_stage_xdl_ngcdhw_gkzyxc_ngkdhw_bf16_pipev5_instance.cpp
+// SPDX-License-Identifier: MIT
+// Copyright (c) 2024, Advanced Micro Devices, Inc. All rights reserved.
+#include "ck/library/tensor_operation_instance/add_device_operation_instance.hpp"
+#include "ck/library/tensor_operation_instance/gpu/grouped_conv_bwd_weight/device_grouped_conv_bwd_weight_two_stage_xdl_instance.hpp"
+namespace ck {
+namespace tensor_operation {
+namespace device {
+namespace instance {
+// Compilation parameters for in[n, hi, wi, g, c] * wei[g, k, y, x, c] = out[n, ho, wo, g, k]
+void add_device_grouped_conv3d_bwd_weight_two_stage_xdl_ngcdhw_gkzyxc_ngkdhw_bf16_pipev5_instances(
+    std::vector<std::unique_ptr<DeviceGroupedConvBwdWeight<3,
+                                                           NGCDHW,
+                                                           GKZYXC,
+                                                           NGKDHW,
+                                                           BF16,
+                                                           BF16,
+                                                           BF16,
+                                                           PassThrough,
+                                                           PassThrough,
+                                                           PassThrough>>>& instances)
+{
+    // 1. Default
+    add_device_operation_instances(
+        instances,
+        device_grouped_conv_bwd_weight_two_stage_ngchw_xdl_c_shuffle_bf16_instances<
+            3,
+            NGCDHW,
+            GKZYXC,
+            NGKDHW,
+            ConvBwdWeightDefault,
+            BlockGemmPipelineScheduler::Intrawave,
+            BlockGemmPipelineVersion::v5>{});
+}
+} // namespace instance
+} // namespace device
+} // namespace tensor_operation
+} // namespace ck
--- a/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/xdl/device_grouped_conv3d_bwd_weight_xdl_gndhwc_gkzyxc_gndhwk_bf16_instance.cpp
+++ b/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/xdl/device_grouped_conv3d_bwd_weight_xdl_gndhwc_gkzyxc_gndhwk_bf16_instance.cpp
 // SPDX-License-Identifier: MIT
-// Copyright (c) 2018-2023, Advanced Micro Devices, Inc. All rights reserved.
+// Copyright (c) 2018-2024, Advanced Micro Devices, Inc. All rights reserved.
 #include "ck/library/tensor_operation_instance/add_device_operation_instance.hpp"
 #include "ck/library/tensor_operation_instance/gpu/grouped_conv_bwd_weight/device_grouped_conv_bwd_weight_xdl_instance.hpp"
@@ -24,19 +24,21 @@ void add_device_grouped_conv3d_bwd_weight_xdl_gndhwc_gkzyxc_gndhwk_bf16_f32_bf16
    // 1. Default
    add_device_operation_instances(
        instances,
-        device_grouped_conv_bwd_weight_xdl_c_shuffle_bf16_instances<3,
+        device_grouped_conv_bwd_weight_xdl_c_shuffle_bf16_f32_bf16_instances<
-                                                                    GNDHWC,
+            3,
-                                                                    GKZYXC,
+            GNDHWC,
-                                                                    GNDHWK,
+            GKZYXC,
-                                                                    ConvBwdWeightDefault>{});
+            GNDHWK,
+            ConvBwdWeightDefault>{});
    // 2. Filter1x1Stride1Pad0
-    add_device_operation_instances(instances,
+    add_device_operation_instances(
-                                   device_grouped_conv_bwd_weight_xdl_c_shuffle_bf16_instances<
+        instances,
-                                       3,
+        device_grouped_conv_bwd_weight_xdl_c_shuffle_bf16_f32_bf16_instances<
-                                       GNDHWC,
+            3,
-                                       GKZYXC,
+            GNDHWC,
-                                       GNDHWK,
+            GKZYXC,
-                                       ConvBwdWeightFilter1x1Stride1Pad0>{});
+            GNDHWK,
+            ConvBwdWeightFilter1x1Stride1Pad0>{});
 }
 } // namespace instance

--- a/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/xdl/device_grouped_conv3d_bwd_weight_xdl_ndhwgc_gkzyxc_ndhwgk_bf16_f32_bf16_instance.cpp
+++ b/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/xdl/device_grouped_conv3d_bwd_weight_xdl_ndhwgc_gkzyxc_ndhwgk_bf16_f32_bf16_instance.cpp
+// SPDX-License-Identifier: MIT
+// Copyright (c) 2024, Advanced Micro Devices, Inc. All rights reserved.
+#include "ck/library/tensor_operation_instance/add_device_operation_instance.hpp"
+#include "ck/library/tensor_operation_instance/gpu/grouped_conv_bwd_weight/device_grouped_conv_bwd_weight_xdl_instance.hpp"
+namespace ck {
+namespace tensor_operation {
+namespace device {
+namespace instance {
+// Compilation parameters for in[n, hi, wi, g, c] * wei[g, k, y, x, c] = out[n, ho, wo, g, k]
+void add_device_grouped_conv3d_bwd_weight_xdl_ndhwgc_gkzyxc_ndhwgk_bf16_f32_bf16_instances(
+    std::vector<std::unique_ptr<DeviceGroupedConvBwdWeight<3,
+                                                           NDHWGC,
+                                                           GKZYXC,
+                                                           NDHWGK,
+                                                           BF16,
+                                                           F32,
+                                                           BF16,
+                                                           PassThrough,
+                                                           PassThrough,
+                                                           PassThrough>>>& instances)
+{
+    // 1. Default
+    add_device_operation_instances(
+        instances,
+        device_grouped_conv_bwd_weight_xdl_c_shuffle_bf16_f32_bf16_instances<
+            3,
+            NDHWGC,
+            GKZYXC,
+            NDHWGK,
+            ConvBwdWeightDefault>{});
+    // 2. Filter1x1Stride1Pad0
+    add_device_operation_instances(
+        instances,
+        device_grouped_conv_bwd_weight_xdl_c_shuffle_bf16_f32_bf16_instances<
+            3,
+            NDHWGC,
+            GKZYXC,
+            NDHWGK,
+            ConvBwdWeightFilter1x1Stride1Pad0>{});
+}
+} // namespace instance
+} // namespace device
+} // namespace tensor_operation
+} // namespace ck
--- a/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/xdl/device_grouped_conv3d_bwd_weight_xdl_ndhwgc_gkzyxc_ndhwgk_bf16_instance.cpp
+++ b/library/src/tensor_operation_instance/gpu/grouped_conv3d_bwd_weight/xdl/device_grouped_conv3d_bwd_weight_xdl_ndhwgc_gkzyxc_ndhwgk_bf16_instance.cpp
 // SPDX-License-Identifier: MIT
-// Copyright (c) 2018-2023, Advanced Micro Devices, Inc. All rights reserved.
+// Copyright (c) 2024, Advanced Micro Devices, Inc. All rights reserved.
 #include "ck/library/tensor_operation_instance/add_device_operation_instance.hpp"
 #include "ck/library/tensor_operation_instance/gpu/grouped_conv_bwd_weight/device_grouped_conv_bwd_weight_xdl_instance.hpp"
@@ -10,13 +10,13 @@ namespace device {
 namespace instance {
 // Compilation parameters for in[n, hi, wi, g, c] * wei[g, k, y, x, c] = out[n, ho, wo, g, k]
-void add_device_grouped_conv3d_bwd_weight_xdl_ndhwgc_gkzyxc_ndhwgk_bf16_f32_bf16_instances(
+void add_device_grouped_conv3d_bwd_weight_xdl_ndhwgc_gkzyxc_ndhwgk_bf16_instances(
    std::vector<std::unique_ptr<DeviceGroupedConvBwdWeight<3,
                                                           NDHWGC,
                                                           GKZYXC,
                                                           NDHWGK,
                                                           BF16,
-                                                           F32,
+                                                           BF16,
                                                           BF16,
                                                           PassThrough,
                                                           PassThrough,

--- a/profiler/src/profile_gemm_universal.cpp
+++ b/profiler/src/profile_gemm_universal.cpp
@@ -57,6 +57,25 @@ int profile_gemm_universal(int argc, char* argv[])
        exit(1);
    }
+    int M;
+    int N;
+    int StrideA;
+    int StrideB;
+    // Analyze the unsupported matrix shapes, switch the M and N number
+    if(std::stoi(argv[9]) % 8 != 0 && std::stoi(argv[8]) % 8 == 0)
+    {
+        M       = std::stoi(argv[9]);
+        StrideA = std::stoi(argv[12]);
+        N       = std::stoi(argv[8]);
+        StrideB = std::stoi(argv[11]);
+    }
+    else
+    {
+        M       = std::stoi(argv[8]);
+        StrideA = std::stoi(argv[11]);
+        N       = std::stoi(argv[9]);
+        StrideB = std::stoi(argv[12]);
+    }
    const auto data_type       = static_cast<GemmDataType>(std::stoi(argv[2]));
    const auto layout          = static_cast<GemmMatrixLayout>(std::stoi(argv[3]));
    const bool do_verification = std::stoi(argv[4]);
@@ -64,12 +83,8 @@ int profile_gemm_universal(int argc, char* argv[])
    const bool do_log          = std::stoi(argv[6]);
    const bool time_kernel     = std::stoi(argv[7]);
-    const int M = std::stoi(argv[8]);
-    const int N = std::stoi(argv[9]);
    const int K = std::stoi(argv[10]);
-    const int StrideA = std::stoi(argv[11]);
-    const int StrideB = std::stoi(argv[12]);
    const int StrideC = std::stoi(argv[13]);
    const int KBatch  = std::stoi(argv[14]);

--- a/profiler/src/profile_grouped_conv_bwd_weight.cpp
+++ b/profiler/src/profile_grouped_conv_bwd_weight.cpp
@@ -25,7 +25,8 @@ enum struct ConvDataType
    F16_F16_F16,        // 1
    BF16_F32_BF16,      // 2
    F16_F16_F16_BF8_F8, // 3
-    I8_I8_I8            // 4
+    I8_I8_I8,           // 4
+    BF16_BF16_BF16,     // 5
 };
 #define OP_NAME "grouped_conv_bwd_weight"
@@ -38,7 +39,8 @@ static void print_helper_msg()
              << "                 1: Input fp16, Weight fp16, Output fp16\n"
              << "                 2: Input bf16, Weight fp32, Output bf16\n"
              << "                 3: Input fp16, Weight fp16, Output fp16, Gemm bf8@fp8\n"
-              << "                 4: Input int8, Weight int8, Output int8)\n"
+              << "                 4: Input int8, Weight int8, Output int8\n"
+              << "                 5: Input bf16, Weight bf16, Output bf16)\n"
              << "arg3: tensor layout (0: Input[G, N, C, Hi, Wi], Weight[G, K, C, Y, X], Output[G, "
                 "N, K, Ho, Wo]\n"
              << "                     1: Input[G, N, Hi, Wi, C], Weight[G, K, Y, X, C], Output[G, "
@@ -180,6 +182,10 @@ int profile_grouped_conv_bwd_weight(int argc, char* argv[])
            // fp32 atomic add is used for weight tensor in bf16 kernel
            return profile(I2, NHWGC{}, GKYXC{}, NHWGK{}, BF16{}, F32{}, BF16{}, BF16{}, BF16{});
        }
+        if(data_type == ConvDataType::BF16_BF16_BF16)
+        {
+            return profile(I2, NHWGC{}, GKYXC{}, NHWGK{}, BF16{}, BF16{}, BF16{}, BF16{}, BF16{});
+        }
    }
    else if(num_dim_spatial == 2 && layout == ConvLayout::NGCHW_GKYXC_NGKHW)
    {
@@ -187,6 +193,11 @@ int profile_grouped_conv_bwd_weight(int argc, char* argv[])
        {
            return profile(I2, NGCHW{}, GKYXC{}, NGKHW{}, F16{}, F16{}, F16{}, F16{}, F16{});
        }
+        if(data_type == ConvDataType::BF16_BF16_BF16)
+        {
+            // fp32 atomic add is used for weight tensor in bf16 kernel
+            return profile(I2, NGCHW{}, GKYXC{}, NGKHW{}, BF16{}, BF16{}, BF16{}, BF16{}, BF16{});
+        }
    }
    if(num_dim_spatial == 3 && layout == ConvLayout::GNHWC_GKYXC_GNHWK)
    {
@@ -224,6 +235,11 @@ int profile_grouped_conv_bwd_weight(int argc, char* argv[])
            // fp32 atomic add is used for weight tensor in bf16 kernel
            return profile(I3, NDHWGC{}, GKZYXC{}, NDHWGK{}, BF16{}, F32{}, BF16{}, BF16{}, BF16{});
        }
+        if(data_type == ConvDataType::BF16_BF16_BF16)
+        {
+            return profile(
+                I3, NDHWGC{}, GKZYXC{}, NDHWGK{}, BF16{}, BF16{}, BF16{}, BF16{}, BF16{});
+        }
        if(data_type == ConvDataType::F16_F16_F16_BF8_F8)
        {
            return profile(I3, NDHWGC{}, GKZYXC{}, NDHWGK{}, F16{}, F16{}, F16{}, BF8{}, F8{});
@@ -240,6 +256,11 @@ int profile_grouped_conv_bwd_weight(int argc, char* argv[])
        {
            return profile(I3, NGCDHW{}, GKZYXC{}, NGKDHW{}, F16{}, F16{}, F16{}, F16{}, F16{});
        }
+        if(data_type == ConvDataType::BF16_BF16_BF16)
+        {
+            return profile(
+                I3, NGCDHW{}, GKZYXC{}, NGKDHW{}, BF16{}, BF16{}, BF16{}, BF16{}, BF16{});
+        }
    }
    std::cout << "this data_type & layout is not implemented" << std::endl;

--- a/script/convert_miopen_driver_to_profiler.py
+++ b/script/convert_miopen_driver_to_profiler.py
@@ -65,8 +65,9 @@ def parse_data_type(args):
        if args.ck_profier_op == "grouped_conv_fwd":
            args.data_type = 3
    if args.data_type == "bfp16":
-        if args.ck_profier_op == "grouped_conv_bwd_weight" or \
+        if args.ck_profier_op == "grouped_conv_bwd_weight":
-           args.ck_profier_op == "grouped_conv_bwd_data" or \
+            args.data_type = 5
+        if args.ck_profier_op == "grouped_conv_bwd_data" or \
           args.ck_profier_op == "grouped_conv_fwd":
            args.data_type = 2

--- a/test/data_type/CMakeLists.txt
+++ b/test/data_type/CMakeLists.txt
@@ -18,4 +18,9 @@ if(result EQUAL 0)
  target_link_libraries(test_bf8 PRIVATE utility)
 endif()
+add_gtest_executable(test_custom_type test_custom_type.cpp)
+if(result EQUAL 0)
+  target_link_libraries(test_custom_type PRIVATE utility)
+endif()
 add_gtest_executable(test_type_convert_const type_convert_const.cpp)
--- a/test/data_type/test_custom_type.cpp
+++ b/test/data_type/test_custom_type.cpp
+// SPDX-License-Identifier: MIT
+// Copyright (c) 2024, Advanced Micro Devices, Inc. All rights reserved.
+#include "gtest/gtest.h"
+#include "ck/utility/data_type.hpp"
+#include "ck/utility/type_convert.hpp"
+using ck::bf8_t;
+using ck::bhalf_t;
+using ck::f8_t;
+using ck::half_t;
+using ck::Number;
+using ck::type_convert;
+using ck::vector_type;
+TEST(Custom_bool, TestSize)
+{
+    struct custom_bool_t
+    {
+        bool data;
+    };
+    ASSERT_EQ(sizeof(custom_bool_t), sizeof(bool));
+    ASSERT_EQ(sizeof(vector_type<custom_bool_t, 2>), sizeof(vector_type<bool, 2>));
+    ASSERT_EQ(sizeof(vector_type<custom_bool_t, 4>), sizeof(vector_type<bool, 4>));
+    ASSERT_EQ(sizeof(vector_type<custom_bool_t, 8>), sizeof(vector_type<bool, 8>));
+    ASSERT_EQ(sizeof(vector_type<custom_bool_t, 16>), sizeof(vector_type<bool, 16>));
+    ASSERT_EQ(sizeof(vector_type<custom_bool_t, 32>), sizeof(vector_type<bool, 32>));
+    ASSERT_EQ(sizeof(vector_type<custom_bool_t, 64>), sizeof(vector_type<bool, 64>));
+}
+TEST(Custom_bool, TestAsType)
+{
+    struct custom_bool_t
+    {
+        using type = bool;
+        type data;
+        custom_bool_t() : data{type{}} {}
+        custom_bool_t(type init) : data{init} {}
+    };
+    // test size
+    const int size             = 4;
+    std::vector<bool> test_vec = {false, true, false, true};
+    // reference vector
+    vector_type<custom_bool_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<custom_bool_t>()(Number<i>{}).data, false);
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_bool_t>()(Number<i>{}) = custom_bool_t{test_vec.at(i)};
+    });
+    // copy the vector
+    vector_type<custom_bool_t, size> left_vec{right_vec};
+    // check if values were copied correctly
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_bool_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_bool, TestAsTypeReshape)
+{
+    struct custom_bool_t
+    {
+        using type = bool;
+        type data;
+        custom_bool_t() : data{type{}} {}
+        custom_bool_t(type init) : data{init} {}
+    };
+    // test size
+    const int size             = 4;
+    std::vector<bool> test_vec = {false, true, false, true};
+    // reference vector
+    vector_type<custom_bool_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<custom_bool_t>()(Number<i>{}).data, false);
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_bool_t>()(Number<i>{}) = custom_bool_t{test_vec.at(i)};
+    });
+    // copy the first half of a vector
+    vector_type<custom_bool_t, size / 2> left_vec{
+        right_vec.template AsType<vector_type<custom_bool_t, size / 2>::type>()(Number<0>{})};
+    // check if values were copied correctly
+    ck::static_for<0, size / 2, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_bool_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_int8, TestSize)
+{
+    struct custom_int8_t
+    {
+        int8_t data;
+    };
+    ASSERT_EQ(sizeof(custom_int8_t), sizeof(int8_t));
+    ASSERT_EQ(sizeof(vector_type<custom_int8_t, 2>), sizeof(vector_type<int8_t, 2>));
+    ASSERT_EQ(sizeof(vector_type<custom_int8_t, 4>), sizeof(vector_type<int8_t, 4>));
+    ASSERT_EQ(sizeof(vector_type<custom_int8_t, 8>), sizeof(vector_type<int8_t, 8>));
+    ASSERT_EQ(sizeof(vector_type<custom_int8_t, 16>), sizeof(vector_type<int8_t, 16>));
+    ASSERT_EQ(sizeof(vector_type<custom_int8_t, 32>), sizeof(vector_type<int8_t, 32>));
+    ASSERT_EQ(sizeof(vector_type<custom_int8_t, 64>), sizeof(vector_type<int8_t, 64>));
+}
+TEST(Custom_int8, TestAsType)
+{
+    struct custom_int8_t
+    {
+        using type = int8_t;
+        type data;
+        custom_int8_t() : data{type{}} {}
+        custom_int8_t(type init) : data{init} {}
+    };
+    // test size
+    const int size               = 4;
+    std::vector<int8_t> test_vec = {3, -6, 8, -2};
+    // reference vector
+    vector_type<custom_int8_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<custom_int8_t>()(Number<i>{}).data, 0);
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_int8_t>()(Number<i>{}) = custom_int8_t{test_vec.at(i)};
+    });
+    // copy the vector
+    vector_type<custom_int8_t, size> left_vec{right_vec};
+    // check if values were copied correctly
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_int8_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_int8, TestAsTypeReshape)
+{
+    struct custom_int8_t
+    {
+        using type = int8_t;
+        type data;
+        custom_int8_t() : data{type{}} {}
+        custom_int8_t(type init) : data{init} {}
+    };
+    // test size
+    const int size               = 4;
+    std::vector<int8_t> test_vec = {3, -6, 8, -2};
+    // reference vector
+    vector_type<custom_int8_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<custom_int8_t>()(Number<i>{}).data, 0);
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_int8_t>()(Number<i>{}) = custom_int8_t{test_vec.at(i)};
+    });
+    // copy the first half of a vector
+    vector_type<custom_int8_t, size / 2> left_vec{
+        right_vec.template AsType<vector_type<custom_int8_t, size / 2>::type>()(Number<0>{})};
+    // check if values were copied correctly
+    ck::static_for<0, size / 2, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_int8_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_uint8, TestSize)
+{
+    struct custom_uint8_t
+    {
+        uint8_t data;
+    };
+    ASSERT_EQ(sizeof(custom_uint8_t), sizeof(uint8_t));
+    ASSERT_EQ(sizeof(vector_type<custom_uint8_t, 2>), sizeof(vector_type<uint8_t, 2>));
+    ASSERT_EQ(sizeof(vector_type<custom_uint8_t, 4>), sizeof(vector_type<uint8_t, 4>));
+    ASSERT_EQ(sizeof(vector_type<custom_uint8_t, 8>), sizeof(vector_type<uint8_t, 8>));
+    ASSERT_EQ(sizeof(vector_type<custom_uint8_t, 16>), sizeof(vector_type<uint8_t, 16>));
+    ASSERT_EQ(sizeof(vector_type<custom_uint8_t, 32>), sizeof(vector_type<uint8_t, 32>));
+    ASSERT_EQ(sizeof(vector_type<custom_uint8_t, 64>), sizeof(vector_type<uint8_t, 64>));
+}
+TEST(Custom_uint8, TestAsType)
+{
+    struct custom_uint8_t
+    {
+        using type = uint8_t;
+        type data;
+        custom_uint8_t() : data{type{}} {}
+        custom_uint8_t(type init) : data{init} {}
+    };
+    // test size
+    const int size                = 4;
+    std::vector<uint8_t> test_vec = {3, 6, 8, 2};
+    // reference vector
+    vector_type<custom_uint8_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<custom_uint8_t>()(Number<i>{}).data, 0);
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_uint8_t>()(Number<i>{}) = custom_uint8_t{test_vec.at(i)};
+    });
+    // copy the vector
+    vector_type<custom_uint8_t, size> left_vec{right_vec};
+    // check if values were copied correctly
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_uint8_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_uint8, TestAsTypeReshape)
+{
+    struct custom_uint8_t
+    {
+        using type = uint8_t;
+        type data;
+        custom_uint8_t() : data{type{}} {}
+        custom_uint8_t(type init) : data{init} {}
+    };
+    // test size
+    const int size                = 4;
+    std::vector<uint8_t> test_vec = {3, 6, 8, 2};
+    // reference vector
+    vector_type<custom_uint8_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<custom_uint8_t>()(Number<i>{}).data, 0);
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_uint8_t>()(Number<i>{}) = custom_uint8_t{test_vec.at(i)};
+    });
+    // copy the first half of a vector
+    vector_type<custom_uint8_t, size / 2> left_vec{
+        right_vec.template AsType<vector_type<custom_uint8_t, size / 2>::type>()(Number<0>{})};
+    // check if values were copied correctly
+    ck::static_for<0, size / 2, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_uint8_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_f8, TestSize)
+{
+    struct custom_f8_t
+    {
+        _BitInt(8) data;
+    };
+    ASSERT_EQ(sizeof(custom_f8_t), sizeof(_BitInt(8)));
+    ASSERT_EQ(sizeof(vector_type<custom_f8_t, 2>), sizeof(vector_type<_BitInt(8), 2>));
+    ASSERT_EQ(sizeof(vector_type<custom_f8_t, 4>), sizeof(vector_type<_BitInt(8), 4>));
+    ASSERT_EQ(sizeof(vector_type<custom_f8_t, 8>), sizeof(vector_type<_BitInt(8), 8>));
+    ASSERT_EQ(sizeof(vector_type<custom_f8_t, 16>), sizeof(vector_type<_BitInt(8), 16>));
+    ASSERT_EQ(sizeof(vector_type<custom_f8_t, 32>), sizeof(vector_type<_BitInt(8), 32>));
+    ASSERT_EQ(sizeof(vector_type<custom_f8_t, 64>), sizeof(vector_type<_BitInt(8), 64>));
+}
+TEST(Custom_f8, TestAsType)
+{
+    struct custom_f8_t
+    {
+        using type = _BitInt(8);
+        type data;
+        custom_f8_t() : data{type{}} {}
+        custom_f8_t(type init) : data{init} {}
+    };
+    // test size
+    const int size                   = 4;
+    std::vector<_BitInt(8)> test_vec = {type_convert<_BitInt(8)>(0.3f),
+                                        type_convert<_BitInt(8)>(-0.6f),
+                                        type_convert<_BitInt(8)>(0.8f),
+                                        type_convert<_BitInt(8)>(-0.2f)};
+    // reference vector
+    vector_type<custom_f8_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}(
+        [&](auto i) { ASSERT_EQ(right_vec.template AsType<custom_f8_t>()(Number<i>{}).data, 0); });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_f8_t>()(Number<i>{}) = custom_f8_t{test_vec.at(i)};
+    });
+    // copy the vector
+    vector_type<custom_f8_t, size> left_vec{right_vec};
+    // check if values were copied correctly
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_f8_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_f8, TestAsTypeReshape)
+{
+    struct custom_f8_t
+    {
+        using type = _BitInt(8);
+        type data;
+        custom_f8_t() : data{type{}} {}
+        custom_f8_t(type init) : data{init} {}
+    };
+    // test size
+    const int size                   = 4;
+    std::vector<_BitInt(8)> test_vec = {type_convert<_BitInt(8)>(0.3f),
+                                        type_convert<_BitInt(8)>(-0.6f),
+                                        type_convert<_BitInt(8)>(0.8f),
+                                        type_convert<_BitInt(8)>(-0.2f)};
+    // reference vector
+    vector_type<custom_f8_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}(
+        [&](auto i) { ASSERT_EQ(right_vec.template AsType<custom_f8_t>()(Number<i>{}).data, 0); });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_f8_t>()(Number<i>{}) = custom_f8_t{test_vec.at(i)};
+    });
+    // copy the first half of a vector
+    vector_type<custom_f8_t, size / 2> left_vec{
+        right_vec.template AsType<vector_type<custom_f8_t, size / 2>::type>()(Number<0>{})};
+    // check if values were copied correctly
+    ck::static_for<0, size / 2, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_f8_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_bf8, TestSize)
+{
+    struct custom_bf8_t
+    {
+        unsigned _BitInt(8) data;
+    };
+    ASSERT_EQ(sizeof(custom_bf8_t), sizeof(unsigned _BitInt(8)));
+    ASSERT_EQ(sizeof(vector_type<custom_bf8_t, 2>), sizeof(vector_type<unsigned _BitInt(8), 2>));
+    ASSERT_EQ(sizeof(vector_type<custom_bf8_t, 4>), sizeof(vector_type<unsigned _BitInt(8), 4>));
+    ASSERT_EQ(sizeof(vector_type<custom_bf8_t, 8>), sizeof(vector_type<unsigned _BitInt(8), 8>));
+    ASSERT_EQ(sizeof(vector_type<custom_bf8_t, 16>), sizeof(vector_type<unsigned _BitInt(8), 16>));
+    ASSERT_EQ(sizeof(vector_type<custom_bf8_t, 32>), sizeof(vector_type<unsigned _BitInt(8), 32>));
+    ASSERT_EQ(sizeof(vector_type<custom_bf8_t, 64>), sizeof(vector_type<unsigned _BitInt(8), 64>));
+}
+TEST(Custom_bf8, TestAsType)
+{
+    struct custom_bf8_t
+    {
+        using type = unsigned _BitInt(8);
+        type data;
+        custom_bf8_t() : data{type{}} {}
+        custom_bf8_t(type init) : data{init} {}
+    };
+    // test size
+    const int size                            = 4;
+    std::vector<unsigned _BitInt(8)> test_vec = {type_convert<unsigned _BitInt(8)>(0.3f),
+                                                 type_convert<unsigned _BitInt(8)>(-0.6f),
+                                                 type_convert<unsigned _BitInt(8)>(0.8f),
+                                                 type_convert<unsigned _BitInt(8)>(-0.2f)};
+    // reference vector
+    vector_type<custom_bf8_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}(
+        [&](auto i) { ASSERT_EQ(right_vec.template AsType<custom_bf8_t>()(Number<i>{}).data, 0); });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_bf8_t>()(Number<i>{}) = custom_bf8_t{test_vec.at(i)};
+    });
+    // copy the vector
+    vector_type<custom_bf8_t, size> left_vec{right_vec};
+    // check if values were copied correctly
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_bf8_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_bf8, TestAsTypeReshape)
+{
+    struct custom_bf8_t
+    {
+        using type = unsigned _BitInt(8);
+        type data;
+        custom_bf8_t() : data{type{}} {}
+        custom_bf8_t(type init) : data{init} {}
+    };
+    // test size
+    const int size                            = 4;
+    std::vector<unsigned _BitInt(8)> test_vec = {type_convert<unsigned _BitInt(8)>(0.3f),
+                                                 type_convert<unsigned _BitInt(8)>(-0.6f),
+                                                 type_convert<unsigned _BitInt(8)>(0.8f),
+                                                 type_convert<unsigned _BitInt(8)>(-0.2f)};
+    // reference vector
+    vector_type<custom_bf8_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}(
+        [&](auto i) { ASSERT_EQ(right_vec.template AsType<custom_bf8_t>()(Number<i>{}).data, 0); });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_bf8_t>()(Number<i>{}) = custom_bf8_t{test_vec.at(i)};
+    });
+    // copy the first half of a vector
+    vector_type<custom_bf8_t, size / 2> left_vec{
+        right_vec.template AsType<vector_type<custom_bf8_t, size / 2>::type>()(Number<0>{})};
+    // check if values were copied correctly
+    ck::static_for<0, size / 2, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_bf8_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_half, TestSize)
+{
+    struct custom_half_t
+    {
+        half_t data;
+    };
+    ASSERT_EQ(sizeof(custom_half_t), sizeof(half_t));
+    ASSERT_EQ(sizeof(vector_type<custom_half_t, 2>), sizeof(vector_type<half_t, 2>));
+    ASSERT_EQ(sizeof(vector_type<custom_half_t, 4>), sizeof(vector_type<half_t, 4>));
+    ASSERT_EQ(sizeof(vector_type<custom_half_t, 8>), sizeof(vector_type<half_t, 8>));
+    ASSERT_EQ(sizeof(vector_type<custom_half_t, 16>), sizeof(vector_type<half_t, 16>));
+    ASSERT_EQ(sizeof(vector_type<custom_half_t, 32>), sizeof(vector_type<half_t, 32>));
+    ASSERT_EQ(sizeof(vector_type<custom_half_t, 64>), sizeof(vector_type<half_t, 64>));
+}
+TEST(Custom_half, TestAsType)
+{
+    struct custom_half_t
+    {
+        using type = half_t;
+        type data;
+        custom_half_t() : data{type{}} {}
+        custom_half_t(type init) : data{init} {}
+    };
+    // test size
+    const int size               = 4;
+    std::vector<half_t> test_vec = {half_t{0.3f}, half_t{-0.6f}, half_t{0.8f}, half_t{-0.2f}};
+    // reference vector
+    vector_type<custom_half_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<custom_half_t>()(Number<i>{}).data,
+                  type_convert<half_t>(0.0f));
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_half_t>()(Number<i>{}) = custom_half_t{test_vec.at(i)};
+    });
+    // copy the vector
+    vector_type<custom_half_t, size> left_vec{right_vec};
+    // check if values were copied correctly
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_half_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_half, TestAsTypeReshape)
+{
+    struct custom_half_t
+    {
+        using type = half_t;
+        type data;
+        custom_half_t() : data{type{}} {}
+        custom_half_t(type init) : data{init} {}
+    };
+    // test size
+    const int size               = 4;
+    std::vector<half_t> test_vec = {half_t{0.3f}, half_t{-0.6f}, half_t{0.8f}, half_t{-0.2f}};
+    // reference vector
+    vector_type<custom_half_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<custom_half_t>()(Number<i>{}).data,
+                  type_convert<half_t>(0.0f));
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_half_t>()(Number<i>{}) = custom_half_t{test_vec.at(i)};
+    });
+    // copy the first half of a vector
+    vector_type<custom_half_t, size / 2> left_vec{
+        right_vec.template AsType<vector_type<custom_half_t, size / 2>::type>()(Number<0>{})};
+    // check if values were copied correctly
+    ck::static_for<0, size / 2, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_half_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_bhalf, TestSize)
+{
+    struct custom_bhalf_t
+    {
+        bhalf_t data;
+    };
+    ASSERT_EQ(sizeof(custom_bhalf_t), sizeof(bhalf_t));
+    ASSERT_EQ(sizeof(vector_type<custom_bhalf_t, 2>), sizeof(vector_type<bhalf_t, 2>));
+    ASSERT_EQ(sizeof(vector_type<custom_bhalf_t, 4>), sizeof(vector_type<bhalf_t, 4>));
+    ASSERT_EQ(sizeof(vector_type<custom_bhalf_t, 8>), sizeof(vector_type<bhalf_t, 8>));
+    ASSERT_EQ(sizeof(vector_type<custom_bhalf_t, 16>), sizeof(vector_type<bhalf_t, 16>));
+    ASSERT_EQ(sizeof(vector_type<custom_bhalf_t, 32>), sizeof(vector_type<bhalf_t, 32>));
+    ASSERT_EQ(sizeof(vector_type<custom_bhalf_t, 64>), sizeof(vector_type<bhalf_t, 64>));
+}
+TEST(Custom_bhalf, TestAsType)
+{
+    struct custom_bhalf_t
+    {
+        using type = bhalf_t;
+        type data;
+        custom_bhalf_t() : data{type{}} {}
+        custom_bhalf_t(type init) : data{init} {}
+    };
+    // test size
+    const int size                = 4;
+    std::vector<bhalf_t> test_vec = {type_convert<bhalf_t>(0.3f),
+                                     type_convert<bhalf_t>(-0.6f),
+                                     type_convert<bhalf_t>(0.8f),
+                                     type_convert<bhalf_t>(-0.2f)};
+    // reference vector
+    vector_type<custom_bhalf_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<custom_bhalf_t>()(Number<i>{}).data,
+                  type_convert<bhalf_t>(0.0f));
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_bhalf_t>()(Number<i>{}) = custom_bhalf_t{test_vec.at(i)};
+    });
+    // copy the vector
+    vector_type<custom_bhalf_t, size> left_vec{right_vec};
+    // check if values were copied correctly
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_bhalf_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_bhalf, TestAsTypeReshape)
+{
+    struct custom_bhalf_t
+    {
+        using type = bhalf_t;
+        type data;
+        custom_bhalf_t() : data{type{}} {}
+        custom_bhalf_t(type init) : data{init} {}
+    };
+    // test size
+    const int size                = 4;
+    std::vector<bhalf_t> test_vec = {type_convert<bhalf_t>(0.3f),
+                                     type_convert<bhalf_t>(-0.6f),
+                                     type_convert<bhalf_t>(0.8f),
+                                     type_convert<bhalf_t>(-0.2f)};
+    // reference vector
+    vector_type<custom_bhalf_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<custom_bhalf_t>()(Number<i>{}).data,
+                  type_convert<bhalf_t>(0.0f));
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_bhalf_t>()(Number<i>{}) = custom_bhalf_t{test_vec.at(i)};
+    });
+    // copy the first half of a vector
+    vector_type<custom_bhalf_t, size / 2> left_vec{
+        right_vec.template AsType<vector_type<custom_bhalf_t, size / 2>::type>()(Number<0>{})};
+    // check if values were copied correctly
+    ck::static_for<0, size / 2, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_bhalf_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_float, TestSize)
+{
+    struct custom_float_t
+    {
+        float data;
+    };
+    ASSERT_EQ(sizeof(custom_float_t), sizeof(float));
+    ASSERT_EQ(sizeof(vector_type<custom_float_t, 2>), sizeof(vector_type<float, 2>));
+    ASSERT_EQ(sizeof(vector_type<custom_float_t, 4>), sizeof(vector_type<float, 4>));
+    ASSERT_EQ(sizeof(vector_type<custom_float_t, 8>), sizeof(vector_type<float, 8>));
+    ASSERT_EQ(sizeof(vector_type<custom_float_t, 16>), sizeof(vector_type<float, 16>));
+    ASSERT_EQ(sizeof(vector_type<custom_float_t, 32>), sizeof(vector_type<float, 32>));
+    ASSERT_EQ(sizeof(vector_type<custom_float_t, 64>), sizeof(vector_type<float, 64>));
+}
+TEST(Custom_float, TestAsType)
+{
+    struct custom_float_t
+    {
+        using type = float;
+        type data;
+        custom_float_t() : data{type{}} {}
+        custom_float_t(type init) : data{init} {}
+    };
+    // test size
+    const int size              = 4;
+    std::vector<float> test_vec = {0.3f, -0.6f, 0.8f, -0.2f};
+    // reference vector
+    vector_type<custom_float_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<custom_float_t>()(Number<i>{}).data, 0.0f);
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_float_t>()(Number<i>{}) = custom_float_t{test_vec.at(i)};
+    });
+    // copy the vector
+    vector_type<custom_float_t, size> left_vec{right_vec};
+    // check if values were copied correctly
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_float_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_float, TestAsTypeReshape)
+{
+    struct custom_float_t
+    {
+        using type = float;
+        type data;
+        custom_float_t() : data{type{}} {}
+        custom_float_t(type init) : data{init} {}
+    };
+    // test size
+    const int size              = 4;
+    std::vector<float> test_vec = {0.3f, -0.6f, 0.8f, -0.2f};
+    // reference vector
+    vector_type<custom_float_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<custom_float_t>()(Number<i>{}).data, 0.0f);
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_float_t>()(Number<i>{}) = custom_float_t{test_vec.at(i)};
+    });
+    // copy the first half of a vector
+    vector_type<custom_float_t, size / 2> left_vec{
+        right_vec.template AsType<vector_type<custom_float_t, size / 2>::type>()(Number<0>{})};
+    // check if values were copied correctly
+    ck::static_for<0, size / 2, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_float_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_double, TestSize)
+{
+    struct custom_double_t
+    {
+        double data;
+    };
+    ASSERT_EQ(sizeof(custom_double_t), sizeof(double));
+    ASSERT_EQ(sizeof(vector_type<custom_double_t, 2>), sizeof(vector_type<double, 2>));
+    ASSERT_EQ(sizeof(vector_type<custom_double_t, 4>), sizeof(vector_type<double, 4>));
+    ASSERT_EQ(sizeof(vector_type<custom_double_t, 8>), sizeof(vector_type<double, 8>));
+    ASSERT_EQ(sizeof(vector_type<custom_double_t, 16>), sizeof(vector_type<double, 16>));
+    ASSERT_EQ(sizeof(vector_type<custom_double_t, 32>), sizeof(vector_type<double, 32>));
+    ASSERT_EQ(sizeof(vector_type<custom_double_t, 64>), sizeof(vector_type<double, 64>));
+}
+TEST(Custom_double, TestAsType)
+{
+    struct custom_double_t
+    {
+        using type = double;
+        type data;
+        custom_double_t() : data{type{}} {}
+        custom_double_t(type init) : data{init} {}
+    };
+    // test size
+    const int size               = 4;
+    std::vector<double> test_vec = {0.3, 0.6, 0.8, 0.2};
+    // reference vector
+    vector_type<custom_double_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<custom_double_t>()(Number<i>{}).data, 0.0);
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_double_t>()(Number<i>{}) = custom_double_t{test_vec.at(i)};
+    });
+    // copy the vector
+    vector_type<custom_double_t, size> left_vec{right_vec};
+    // check if values were copied correctly
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_double_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Custom_double, TestAsTypeReshape)
+{
+    struct custom_double_t
+    {
+        using type = double;
+        type data;
+        custom_double_t() : data{type{}} {}
+        custom_double_t(type init) : data{init} {}
+    };
+    // test size
+    const int size               = 4;
+    std::vector<double> test_vec = {0.3, 0.6, 0.8, 0.2};
+    // reference vector
+    vector_type<custom_double_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<custom_double_t>()(Number<i>{}).data, 0.0);
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<custom_double_t>()(Number<i>{}) = custom_double_t{test_vec.at(i)};
+    });
+    // copy the first half of a vector
+    vector_type<custom_double_t, size / 2> left_vec{
+        right_vec.template AsType<vector_type<custom_double_t, size / 2>::type>()(Number<0>{})};
+    // check if values were copied correctly
+    ck::static_for<0, size / 2, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<custom_double_t>()(Number<i>{}).data, test_vec.at(i));
+    });
+}
+TEST(Complex_half, TestSize)
+{
+    struct complex_half_t
+    {
+        half_t real;
+        half_t img;
+    };
+    ASSERT_EQ(sizeof(complex_half_t), sizeof(half_t) + sizeof(half_t));
+    ASSERT_EQ(sizeof(vector_type<complex_half_t, 2>),
+              sizeof(vector_type<half_t, 2>) + sizeof(vector_type<half_t, 2>));
+    ASSERT_EQ(sizeof(vector_type<complex_half_t, 4>),
+              sizeof(vector_type<half_t, 4>) + sizeof(vector_type<half_t, 4>));
+    ASSERT_EQ(sizeof(vector_type<complex_half_t, 8>),
+              sizeof(vector_type<half_t, 8>) + sizeof(vector_type<half_t, 8>));
+    ASSERT_EQ(sizeof(vector_type<complex_half_t, 16>),
+              sizeof(vector_type<half_t, 16>) + sizeof(vector_type<half_t, 16>));
+    ASSERT_EQ(sizeof(vector_type<complex_half_t, 32>),
+              sizeof(vector_type<half_t, 32>) + sizeof(vector_type<half_t, 32>));
+    ASSERT_EQ(sizeof(vector_type<complex_half_t, 64>),
+              sizeof(vector_type<half_t, 64>) + sizeof(vector_type<half_t, 64>));
+}
+TEST(Complex_half, TestAlignment)
+{
+    struct complex_half_t
+    {
+        half_t real;
+        half_t img;
+    };
+    ASSERT_EQ(alignof(vector_type<complex_half_t, 2>),
+              alignof(vector_type<half_t, 2>) + alignof(vector_type<half_t, 2>));
+    ASSERT_EQ(alignof(vector_type<complex_half_t, 4>),
+              alignof(vector_type<half_t, 4>) + alignof(vector_type<half_t, 4>));
+    ASSERT_EQ(alignof(vector_type<complex_half_t, 8>),
+              alignof(vector_type<half_t, 8>) + alignof(vector_type<half_t, 8>));
+    ASSERT_EQ(alignof(vector_type<complex_half_t, 16>),
+              alignof(vector_type<half_t, 16>) + alignof(vector_type<half_t, 16>));
+    ASSERT_EQ(alignof(vector_type<complex_half_t, 32>),
+              alignof(vector_type<half_t, 32>) + alignof(vector_type<half_t, 32>));
+    ASSERT_EQ(alignof(vector_type<complex_half_t, 64>),
+              alignof(vector_type<half_t, 64>) + alignof(vector_type<half_t, 64>));
+}
+TEST(Complex_half, TestAsType)
+{
+    struct complex_half_t
+    {
+        using type = half_t;
+        type real;
+        type img;
+        complex_half_t() : real{type{}}, img{type{}} {}
+        complex_half_t(type real_init, type img_init) : real{real_init}, img{img_init} {}
+    };
+    // test size
+    const int size = 4;
+    // custom type number of elements
+    const int num_elem           = sizeof(complex_half_t) / sizeof(complex_half_t::type);
+    std::vector<half_t> test_vec = {half_t{0.3f},
+                                    half_t{-0.6f},
+                                    half_t{0.8f},
+                                    half_t{-0.2f},
+                                    half_t{0.5f},
+                                    half_t{-0.7f},
+                                    half_t{0.9f},
+                                    half_t{-0.3f}};
+    // reference vector
+    vector_type<complex_half_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<complex_half_t>()(Number<i>{}).real,
+                  type_convert<half_t>(0.0f));
+        ASSERT_EQ(right_vec.template AsType<complex_half_t>()(Number<i>{}).img,
+                  type_convert<half_t>(0.0f));
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<complex_half_t>()(Number<i>{}) =
+            complex_half_t{test_vec.at(num_elem * i), test_vec.at(num_elem * i + 1)};
+    });
+    // copy the vector
+    vector_type<complex_half_t, size> left_vec{right_vec};
+    // check if values were copied correctly
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<complex_half_t>()(Number<i>{}).real,
+                  test_vec.at(num_elem * i));
+        ASSERT_EQ(left_vec.template AsType<complex_half_t>()(Number<i>{}).img,
+                  test_vec.at(num_elem * i + 1));
+    });
+}
+TEST(Complex_half, TestAsTypeReshape)
+{
+    struct complex_half_t
+    {
+        using type = half_t;
+        type real;
+        type img;
+        complex_half_t() : real{type{}}, img{type{}} {}
+        complex_half_t(type real_init, type img_init) : real{real_init}, img{img_init} {}
+    };
+    // test size
+    const int size = 4;
+    // custom type number of elements
+    const int num_elem           = sizeof(complex_half_t) / sizeof(complex_half_t::type);
+    std::vector<half_t> test_vec = {half_t{0.3f},
+                                    half_t{-0.6f},
+                                    half_t{0.8f},
+                                    half_t{-0.2f},
+                                    half_t{0.5f},
+                                    half_t{-0.7f},
+                                    half_t{0.9f},
+                                    half_t{-0.3f}};
+    // reference vector
+    vector_type<complex_half_t, size> right_vec;
+    // check default CTOR
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        ASSERT_EQ(right_vec.template AsType<complex_half_t>()(Number<i>{}).real,
+                  type_convert<half_t>(0.0f));
+        ASSERT_EQ(right_vec.template AsType<complex_half_t>()(Number<i>{}).img,
+                  type_convert<half_t>(0.0f));
+    });
+    // assign test values to the vector
+    ck::static_for<0, size, 1>{}([&](auto i) {
+        right_vec.template AsType<complex_half_t>()(Number<i>{}) =
+            complex_half_t{test_vec.at(num_elem * i), test_vec.at(num_elem * i + 1)};
+    });
+    // copy the first half of a vector
+    vector_type<complex_half_t, size / 2> left_vec{
+        right_vec.template AsType<vector_type<complex_half_t, size / 2>::type>()(Number<0>{})};
+    // check if values were copied correctly
+    ck::static_for<0, size / 2, 1>{}([&](auto i) {
+        ASSERT_EQ(left_vec.template AsType<complex_half_t>()(Number<i>{}).real,
+                  test_vec.at(num_elem * i));
+        ASSERT_EQ(left_vec.template AsType<complex_half_t>()(Number<i>{}).img,
+                  test_vec.at(num_elem * i + 1));
+    });
+}