Merge branch 'develop' into barkocot/strides-for-weights-conv

23220fe5 · zjing14 · GitHub · 7d6fa69a · 2474dddb · 23220fe5
Unverified Commit 23220fe5 authored Aug 03, 2023 by zjing14 Committed by GitHub Aug 03, 2023
11 changed files
--- a/CMakeLists.txt
+++ b/CMakeLists.txt
@@ -60,11 +60,34 @@ message("checking which targets are supported")
 #This is the list of targets to be used in case GPU_TARGETS is not set on command line
 #These targets will be filtered and only supported ones will be used
 #Setting GPU_TARGETS on command line will override this list
-rocm_check_target_ids(DEFAULT_GPU_TARGETS
+if(NOT PROFILER_ONLY)
-    TARGETS "gfx900;gfx906;gfx908;gfx90a;gfx940;gfx941;gfx942;gfx1030;gfx1100;gfx1101;gfx1102"
+    rocm_check_target_ids(DEFAULT_GPU_TARGETS
-)
+        TARGETS "gfx900;gfx906;gfx908;gfx90a;gfx940;gfx941;gfx942;gfx1030;gfx1100;gfx1101;gfx1102")
+else()
+    add_definitions(-DPROFILER_ONLY)
+    if(GPU_TARGETS)
+        message(FATAL_ERROR "For PROFILE_ONLY build, please do not set GPU_TARGETS, use GPU_ARCH = gfx9, gfx10, or gfx11")
+    endif()
+    if(GPU_ARCH MATCHES "gfx9")
+        rocm_check_target_ids(DEFAULT_GPU_TARGETS TARGETS "gfx900;gfx906;gfx908;gfx90a;gfx940;gfx941;gfx942")
+    elseif(GPU_ARCH MATCHES "gfx10")
+        rocm_check_target_ids(DEFAULT_GPU_TARGETS TARGETS "gfx1030")
+    elseif(GPU_ARCH MATCHES "gfx11")
+        rocm_check_target_ids(DEFAULT_GPU_TARGETS TARGETS "gfx1100;gfx1101;gfx1102")
+    else()
+        message(FATAL_ERROR "For PROFILE_ONLY build, please specify GPU_ARCH as gfx9, gfx10, or gfx11")
+    endif()
+endif()
 message("Supported GPU_TARGETS= ${DEFAULT_GPU_TARGETS}")
 set(AMDGPU_TARGETS "${DEFAULT_GPU_TARGETS}" CACHE STRING " ")
+if(GPU_TARGETS)
+    message("Building CK for the following targets: ${GPU_TARGETS}")
+else()
+    message("Building CK for the following targets: ${AMDGPU_TARGETS}")
+endif()
 find_package(hip)
 option(USE_BITINT_EXTENSION_INT4, "Whether to enable clang's BitInt extension to provide int4 data type." OFF)
@@ -347,6 +370,7 @@ add_custom_target(instances DEPENDS utility;${CK_DEVICE_INSTANCES}  SOURCES ${IN
 add_subdirectory(library)
 if(NOT DEFINED INSTANCES_ONLY)
+ if(NOT DEFINED PROFILER_ONLY)
   rocm_package_setup_component(tests
        LIBRARY_NAME composablekernel
        PACKAGE_NAME tests # Prevent -static suffix on package name
@@ -356,15 +380,22 @@ if(NOT DEFINED INSTANCES_ONLY)
        LIBRARY_NAME composablekernel
        PACKAGE_NAME examples
   )
+   add_subdirectory(example)
+   add_subdirectory(test)
   rocm_package_setup_component(profiler
        LIBRARY_NAME composablekernel
        PACKAGE_NAME ckProfiler
   )
-   add_subdirectory(example)
-   add_subdirectory(test)
   add_subdirectory(profiler)
+  else()
+    #When building PROFILER_ONLY, label the package with GPU_ARCH
+    rocm_package_setup_component(profiler
+       LIBRARY_NAME composablekernel
+       PACKAGE_NAME ckProfiler_${GPU_ARCH}
+    )
+    add_subdirectory(profiler)
+  endif()
 endif()
 #Create an interface target for the include only files and call it "composablekernels"

--- a/docs/API_Reference_Guide.rst
+++ b/docs/API_Reference_Guide.rst
@@ -7,8 +7,8 @@ API Reference Guide
 Introduction
 =================
-This document contains details of the APIs for the Composable Kernel (CK) library and introduces some of the key design
+This document contains details of the APIs for the Composable Kernel (CK) library and introduces
-principles that are used to write new classes that extend CK functionality.
+some of the key design principles that are used to write new classes that extend CK functionality.
 =================
 Using CK API
@@ -30,8 +30,8 @@ DeviceMem
 Kernels For Flashattention
 ---------------------------
-The Flashattention algorithm is defined in :cite:t:`dao2022flashattention`.  This sections lists the classes that are
+The Flashattention algorithm is defined in :cite:t:`dao2022flashattention`. This sections lists
-used in the CK GPU implementation of Flashattention.
+the classes that are used in the CK GPU implementation of Flashattention.
 **Gridwise classes**

--- a/docs/Supported_Primitives_Guide.rst
+++ b/docs/Supported_Primitives_Guide.rst
@@ -2,15 +2,16 @@
 Supported Primitives Guide
 ==========================
-This document contains details of supported primitives in Composable Kernel (CK). In contrast to the API Reference
+This document contains details of supported primitives in Composable Kernel (CK). In contrast to the
-Guide, the Supported Primitives Guide is an introduction to the math which underpins the algorithms implemented in CK.
+API Reference Guide, the Supported Primitives Guide is an introduction to the math which underpins
+the algorithms implemented in CK.
 ------------
 Softmax
 ------------
-For vectors :math:`x^{(1)}, x^{(2)}, \ldots, x^{(T)}` of size :math:`B` we can decompose the softmax of concatenated
+For vectors :math:`x^{(1)}, x^{(2)}, \ldots, x^{(T)}` of size :math:`B` we can decompose the
-:math:`x = [ x^{(1)}\ | \ \ldots \ | \ x^{(T)} ]` as,
+softmax of concatenated :math:`x = [ x^{(1)}\ | \ \ldots \ | \ x^{(T)} ]` as,
 .. math::
   :nowrap:
@@ -25,8 +26,8 @@ For vectors :math:`x^{(1)}, x^{(2)}, \ldots, x^{(T)}` of size :math:`B` we can d
 where :math:`f(x^{(j)}) = \exp( x^{(j)} - m(x^{(j)}) )` is of size :math:`B` and
 :math:`z(x^{(j)}) = f(x_1^{(j)})+ \ldots+ f(x_B^{(j)})` is a scalar.
-For a matrix :math:`X` composed of :math:`T_r \times T_c` tiles, :math:`X_{ij}`, of size :math:`B_r \times B_c` we can
+For a matrix :math:`X` composed of :math:`T_r \times T_c` tiles, :math:`X_{ij}`, of size
-compute the row-wise softmax as follows.
+:math:`B_r \times B_c` we can compute the row-wise softmax as follows.
 For :math:`j` from :math:`1` to :math:`T_c`, and :math:`i` from :math:`1` to :math:`T_r` calculate,

--- a/docs/dockerhub.rst
+++ b/docs/dockerhub.rst
 ===================
-CK docker hub
+CK Docker Hub
 ===================
-`Docker hub <https://hub.docker.com/r/rocm/composable_kernel>`_
 -------------------------------------
 Why do I need this?
 -------------------------------------
-To make our lives easier and bring Composable Kernel dependencies together, we recommend using docker images.
+To make our lives easier and bring Composable Kernel dependencies together, we recommend using
+docker images that can be found on `Docker Hub <https://hub.docker.com/r/rocm/composable_kernel>`_.
 -------------------------------------
 So what is Composable Kernel?
 -------------------------------------
-Composable Kernel (CK) library aims to provide a programming model for writing performance critical kernels for machine learning workloads across multiple architectures including GPUs, CPUs, etc, through general purpose kernel languages, like HIP C++.
+Composable Kernel (CK) library aims to provide a programming model for writing performance critical
+kernels for machine learning workloads across multiple architectures including GPUs, CPUs, etc,
+through general purpose kernel languages, like HIP C++.
 To get the CK library::
    git clone https://github.com/ROCmSoftwarePlatform/composable_kernel.git
 run a docker container::
    docker run                                                            \
@@ -30,7 +30,7 @@ run a docker container::
    --group-add sudo                                                      \
    -w /root/workspace                                                    \
    -v ${PATH_TO_LOCAL_WORKSPACE}:/root/workspace                         \
-    rocm/composable_kernel:ck_ub20.04_rocm5.3_release                     \
+    rocm/composable_kernel:ck_ub20.04_rocm5.6                             \
    /bin/bash
 and build the CK::
@@ -58,7 +58,9 @@ We can also run specific examples or tests like::
    ./bin/example_gemm_xdl_fp16
    ./bin/test_gemm_fp16
-For more details visit `CK github repo <https://github.com/ROCmSoftwarePlatform/composable_kernel>`_, `CK examples <https://github.com/ROCmSoftwarePlatform/composable_kernel/tree/develop/example)>`_, `even more CK examples <https://github.com/ROCmSoftwarePlatform/composable_kernel/tree/develop/client_example>`_.
+For more details visit `CK github repository <https://github.com/ROCmSoftwarePlatform/composable_kernel>`_,
+`CK examples <https://github.com/ROCmSoftwarePlatform/composable_kernel/tree/develop/example)>`_,
+`even more CK examples <https://github.com/ROCmSoftwarePlatform/composable_kernel/tree/develop/client_example>`_.
 -------------------------------------
 And what is inside?
@@ -74,12 +76,11 @@ The docker images have everything you need for running CK including:
 Which image is right for me?
 -------------------------------------
-Let's take a look at the image naming, for example "ck_ub20.04_rocm5.4_release". The image specs are:
+Let's take a look at the image naming, for example ``ck_ub20.04_rocm5.6``. The image specs are:
-* "ck" - made for running Composable Kernel
+* ``ck`` - made for running Composable Kernel;
-* "ub20.04" - based on Ubuntu 20.04
+* ``ub20.04`` - based on Ubuntu 20.04;
-* "rocm5.4" - ROCm platform version 5.4
+* ``rocm5.6`` - ROCm platform version 5.6.
-* "release" - compiler version is release
 So just pick the right image for your project dependencies and you're all set.
@@ -87,7 +88,9 @@ So just pick the right image for your project dependencies and you're all set.
 DIY starts here
 -------------------------------------
-If you need to customize a docker image or just can't stop tinkering, feel free to adjust the `Dockerfile <https://github.com/ROCmSoftwarePlatform/composable_kernel/blob/develop/Dockerfile>`_ for your needs.
+If you need to customize a docker image or just can't stop tinkering, feel free to adjust the
+`Dockerfile <https://github.com/ROCmSoftwarePlatform/composable_kernel/blob/develop/Dockerfile>`_
+for your needs.
 -------------------------------------
 License

--- a/docs/index.rst
+++ b/docs/index.rst
@@ -12,12 +12,15 @@ This document contains instructions for installing, using, and contributing to C
 Methodology
 -----------
-Composable Kernel (CK) library aims to provide a programming model for writing performance critical kernels for machine learning workloads across multiple architectures including GPUs, CPUs, etc, through general purpose kernel languages, like HIP C++.
+Composable Kernel (CK) library aims to provide a programming model for writing performance critical
+kernels for machine learning workloads across multiple architectures including GPUs, CPUs, etc,
+through general purpose kernel languages, like HIP C++.
 CK utilizes two concepts to achieve performance portability and code maintainability:
 * A tile-based programming model
-* Algorithm complexity reduction for complex ML operators, using innovative technique we call "Tensor Coordinate Transformation".
+* Algorithm complexity reduction for complex ML operators, using innovative technique we call
+  "Tensor Coordinate Transformation".
 .. image:: data/ck_component.png
   :alt: CK Components

--- a/docs/tutorial_hello_world.rst
+++ b/docs/tutorial_hello_world.rst
@@ -6,15 +6,26 @@ CK Hello world
 Motivation
 -------------------------------------
-This tutorial is aimed at engineers dealing with artificial intelligence and machine learning who would like to optimize their pipelines and squeeze every performance drop by adding Composable Kernel (CK) library to their projects. We would like to make the CK library approachable so the tutorial is not based on the latest release and doesn't have all the bleeding edge features, but it will be reproducible now and forever.
+This tutorial is aimed at engineers dealing with artificial intelligence and machine learning who
+would like to optimize their pipelines and squeeze every performance drop by adding Composable
+Kernel (CK) library to their projects. We would like to make the CK library approachable so
+the tutorial is not based on the latest release and doesn't have all the bleeding edge features,
+but it will be reproducible now and forever.
-During this tutorial we will have an introduction to the CK library, we will build it and run some examples and tests, so to say we will run a "Hello world" example. In future tutorials we will go in depth and breadth and get familiar with other tools and ways to integrate CK into your project.
+During this tutorial we will have an introduction to the CK library, we will build it and run some
+examples and tests, so to say we will run a "Hello world" example. In future tutorials we will go
+in depth and breadth and get familiar with other tools and ways to integrate CK into your project.
 -------------------------------------
 Description
 -------------------------------------
-Modern AI technology solves more and more problems in all imaginable fields, but crafting fast and efficient workflows is still challenging. CK is one of the tools to make AI heavy lifting as fast and efficient as possible. CK is a collection of optimized AI operator kernels and tools to create new ones. The library has components required for majority of modern neural networks architectures including matrix multiplication, convolution, contraction, reduction, attention modules, variety of activation functions, fused operators and many more.
+Modern AI technology solves more and more problems in all imaginable fields, but crafting fast and
+efficient workflows is still challenging. CK is one of the tools to make AI heavy lifting as fast
+and efficient as possible. CK is a collection of optimized AI operator kernels and tools to create
+new ones. The library has components required for majority of modern neural networks architectures
+including matrix multiplication, convolution, contraction, reduction, attention modules, variety of
+activation functions, fused operators and many more.
 So how do we (almost) reach the speed of light? CK acceleration abilities are based on:
@@ -24,15 +35,18 @@ So how do we (almost) reach the speed of light? CK acceleration abilities are ba
 * Hardware acceleration use.
 * Support of low precision data types including fp16, bf16, int8 and int4.
-If you are excited and need more technical details and benchmarking results - read this awesome `blog post <https://community.amd.com/t5/instinct-accelerators/amd-composable-kernel-library-efficient-fused-kernels-for-ai/ba-p/553224>`_.
+If you are excited and need more technical details and benchmarking results - read this awesome
+`blog post <https://community.amd.com/t5/instinct-accelerators/amd-composable-kernel-library-efficient-fused-kernels-for-ai/ba-p/553224>`_.
-For more details visit our `github repo <https://github.com/ROCmSoftwarePlatform/composable_kernel>`_.
+For more details visit our `github repository <https://github.com/ROCmSoftwarePlatform/composable_kernel>`_.
 -------------------------------------
 Hardware targets
 -------------------------------------
-CK library fully supports "gfx908" and "gfx90a" GPU architectures and only some operators are supported for "gfx1030". Let's check the hardware you have at hand and decide on the target GPU architecture
+CK library fully supports `gfx908` and `gfx90a` GPU architectures and only some operators are
+supported for `gfx1030`. Let's check the hardware you have at hand and decide on the target
+GPU architecture.
 ==========     =========
 GPU Target     AMD GPU
@@ -42,7 +56,8 @@ gfx90a 	       Radeon Instinct MI210, MI250, MI250X
 gfx1030        Radeon PRO V620, W6800, W6800X, W6800X Duo, W6900X, RX 6800, RX 6800 XT, RX 6900 XT, RX 6900 XTX, RX 6950 XT
 ==========     =========
-There are also `cloud options <https://aws.amazon.com/ec2/instance-types/g4/>`_ you can find if you don't have an AMD GPU at hand.
+There are also `cloud options <https://aws.amazon.com/ec2/instance-types/g4/>`_ you can find if
+you don't have an AMD GPU at hand.
 -------------------------------------
 Build the library
@@ -54,9 +69,13 @@ First let's clone the library and rebase to the tested version::
    cd composable_kernel/
    git checkout tutorial_hello_world
-To make our lives easier we prepared `docker images <https://hub.docker.com/r/rocm/composable_kernel>`_ with all the necessary dependencies. Pick the right image and create a container. In this tutorial we use "rocm/composable_kernel:ck_ub20.04_rocm5.3_release" image, it is based on Ubuntu 20.04, ROCm v5.3, compiler release version.
+To make our lives easier we prepared
+`docker images <https://hub.docker.com/r/rocm/composable_kernel>`_ with all the necessary
+dependencies. Pick the right image and create a container. In this tutorial we use
+``rocm/composable_kernel:ck_ub20.04_rocm5.6`` image, it is based on Ubuntu 20.04 and
+ROCm v5.6.
-If your current folder is ${HOME}, start the docker container with::
+If your current folder is ``${HOME}``, start the docker container with::
    docker run  \
    -it  \
@@ -64,20 +83,23 @@ If your current folder is ${HOME}, start the docker container with::
    --group-add sudo  \
    -w /root/workspace  \
    -v ${HOME}:/root/workspace  \
-    rocm/composable_kernel:ck_ub20.04_rocm5.3_release  \
+    rocm/composable_kernel:ck_ub20.04_rocm5.6  \
    /bin/bash
-If your current folder is different from ${HOME}, adjust the line `-v ${HOME}:/root/workspace` to fit your folder structure.
+If your current folder is different from ``${HOME}``, adjust the line ``-v ${HOME}:/root/workspace``
+to fit your folder structure.
-Inside the docker container current folder is "~/workspace", library path is "~/workspace/composable_kernel", navigate to the library::
+Inside the docker container current folder is ``~/workspace``, library path is
+``~/workspace/composable_kernel``, navigate to the library::
    cd composable_kernel/
-Create and go to the "build" directory::
+Create and go to the ``build`` directory::
    mkdir build && cd build
-In the previous section we talked about target GPU architecture. Once you decide which one is right for you, run cmake using the right GPU_TARGETS flag::
+In the previous section we talked about target GPU architecture. Once you decide which one is right
+for you, run CMake using the right ``GPU_TARGETS`` flag::
    cmake  \
    -D CMAKE_PREFIX_PATH=/opt/rocm  \
@@ -87,7 +109,7 @@ In the previous section we talked about target GPU architecture. Once you decide
    -D BUILD_DEV=OFF  \
    -D GPU_TARGETS="gfx908;gfx90a;gfx1030" ..
-If everything went well the cmake run will end up with::
+If everything went well the CMake run will end up with::
    -- Configuring done
    -- Generating done
@@ -118,9 +140,12 @@ We can also run them separately, here is a separate example execution::
    ./bin/example_gemm_xdl_fp16 1 1 1
-The arguments "1 1 1" mean that we want to run this example in the mode: verify results with CPU, initialize matrices with integers and benchmark the kernel execution. You can play around with these parameters and see how output and execution results change.
+The arguments ``1 1 1`` mean that we want to run this example in the mode: verify results with CPU,
+initialize matrices with integers and benchmark the kernel execution. You can play around with
+these parameters and see how output and execution results change.
-If everything goes well and you have a device based on gfx908 or gfx90a architecture you should see something like::
+If everything goes well and you have a device based on `gfx908` or `gfx90a` architecture you should see
+something like::
    a_m_k: dim 2, lengths {3840, 4096}, strides {4096, 1}
    b_k_n: dim 2, lengths {4096, 4096}, strides {1, 4096}
@@ -130,14 +155,15 @@ If everything goes well and you have a device based on gfx908 or gfx90a architec
    Start running 10 times...
    Perf: 1.10017 ms, 117.117 TFlops, 87.6854 GB/s, DeviceGemmXdl<256, 256, 128, 4, 8, 32, 32, 4, 2> NumPrefetch: 1, LoopScheduler: Default, PipelineVersion: v1
-Meanwhile, running it on a gfx1030 device should result in::
+Meanwhile, running it on a `gfx1030` device should result in::
    a_m_k: dim 2, lengths {3840, 4096}, strides {4096, 1}
    b_k_n: dim 2, lengths {4096, 4096}, strides {1, 4096}
    c_m_n: dim 2, lengths {3840, 4096}, strides {4096, 1}
    DeviceGemmXdl<256, 256, 128, 4, 8, 32, 32, 4, 2> NumPrefetch: 1, LoopScheduler: Default, PipelineVersion: v1 does not support this problem
-But don't panic, some of the operators are supported on gfx1030 architecture, so you can run a separate example like::
+But don't panic, some of the operators are supported on `gfx1030` architecture, so you can run a
+separate example like::
    ./bin/example_gemm_dl_fp16 1 1 1
@@ -154,7 +180,14 @@ and it should result in something nice similar to::
    Start running 10 times...
    Perf: 3.65695 ms, 35.234 TFlops, 26.3797 GB/s, DeviceGemmDl<256, 128, 128, 16, 2, 4, 4, 1>
-Or we can run a separate test::
+.. note::
+    There was a new CMake flag ``DL_KERNELS`` added in the latest versions of CK. If you use one of
+    the newest versions of the library and do not see the above results when running
+    ``example_gemm_dl_fp16``, it might be necessary to add ``-D DL_KERNELS=ON`` to your CMake command
+    in order to build the operators supported on the `gfx1030` architecture.
+We can also run a separate test::
    ctest -R test_gemm_fp16
@@ -169,6 +202,9 @@ If everything goes well you should see something like::
 Summary
 -----------
-In this tutorial we took the first look at the Composable Kernel library, built it on your system and ran some examples and tests. Stay tuned, in the next tutorial we will run kernels with different configs to find out the best one for your hardware and task.
+In this tutorial we took the first look at the Composable Kernel library, built it on your system
+and ran some examples and tests. Stay tuned, in the next tutorial we will run kernels with different
+configs to find out the best one for your hardware and task.
-P.S.: Don't forget to switch out the cloud instance if you have launched one, you can find better ways to spend your money for sure!
+P.S.: Don't forget to switch off the cloud instance if you have launched one, you can find better
+ways to spend your money for sure!
--- a/include/ck/ck.hpp
+++ b/include/ck/ck.hpp
@@ -198,7 +198,7 @@
 #define CK_WORKAROUND_SWDEV_388832 1
 // workaround: Grouped Conv2d_bwd_data fails for already implemented instance
-#define CK_WORKAROUND_SWDEV_3318619 0
+#define CK_WORKAROUND_GITHUB_ISSUE_824 1
 // flag to enable (1) or disable (0) the debugging output in some kernels
 #define DEBUG_LOG 0

--- a/library/include/ck/library/tensor_operation_instance/gpu/grouped_conv_bwd_data/device_grouped_conv_bwd_data_xdl_instance.hpp
+++ b/library/include/ck/library/tensor_operation_instance/gpu/grouped_conv_bwd_data/device_grouped_conv_bwd_data_xdl_instance.hpp
--- a/library/src/tensor_operation_instance/gpu/gemm/CMakeLists.txt
+++ b/library/src/tensor_operation_instance/gpu/gemm/CMakeLists.txt
@@ -95,36 +95,39 @@ endif()
 add_instance_library(device_gemm_instance ${GEMM_INSTANCES})
-set(ENABLE_PIPELINE_V2_OPT OFF)
-if (ENABLE_PIPELINE_V2_OPT)
+if(DTYPES MATCHES "fp16" OR NOT DEFINED DTYPES)
-set(MAX_ILP_OPTS
+  set(ENABLE_PIPELINE_V2_OPT OFF)
-   -mllvm
-   -amdgpu-enable-max-ilp-scheduling-strategy
+  if (ENABLE_PIPELINE_V2_OPT)
-)
+    set(MAX_ILP_OPTS
-set(WAVES_PER_EU_DEFS
+       -mllvm
-   CK_USE_WAVES_PER_EU=1
+       -amdgpu-enable-max-ilp-scheduling-strategy
-   CK_MIN_WAVES_PER_EU=1
+    )
-   CK_MAX_WAVES_PER_EU=1
+    set(WAVES_PER_EU_DEFS
-)
+       CK_USE_WAVES_PER_EU=1
-set(IGLP_OPT_DEFS
+       CK_MIN_WAVES_PER_EU=1
-   CK_EXPERIMENTAL_PIPELINE_V2_IGLP_OPT=1
+       CK_MAX_WAVES_PER_EU=1
-)
+    )
+    set(IGLP_OPT_DEFS
+       CK_EXPERIMENTAL_PIPELINE_V2_IGLP_OPT=1
+    )
-# layout=NT
+    # layout=NT
-set_source_files_properties(device_gemm_xdl_f16_f16_f16/km_kn_mn_default_pipeline_v2_opt_instance.cpp PROPERTIES
+    set_source_files_properties(device_gemm_xdl_f16_f16_f16/km_kn_mn_default_pipeline_v2_opt_instance.cpp PROPERTIES
-   COMPILE_OPTIONS ";;"
+       COMPILE_OPTIONS ";;"
-   COMPILE_DEFINITIONS "${WAVES_PER_EU_DEFS};${IGLP_OPT_DEFS}")
+       COMPILE_DEFINITIONS "${WAVES_PER_EU_DEFS};${IGLP_OPT_DEFS}")
-# layout=NN
+    # layout=NN
-set_source_files_properties(device_gemm_xdl_f16_f16_f16/km_nk_mn_default_pipeline_v2_opt_instance.cpp PROPERTIES
+    set_source_files_properties(device_gemm_xdl_f16_f16_f16/km_nk_mn_default_pipeline_v2_opt_instance.cpp PROPERTIES
-   COMPILE_OPTIONS "${MAX_ILP_OPTS}"
+       COMPILE_OPTIONS "${MAX_ILP_OPTS}"
-   COMPILE_DEFINITIONS "${WAVES_PER_EU_DEFS};${IGLP_OPT_DEFS}")
+       COMPILE_DEFINITIONS "${WAVES_PER_EU_DEFS};${IGLP_OPT_DEFS}")
-# layout=TT
+    # layout=TT
-set_source_files_properties(device_gemm_xdl_f16_f16_f16/mk_kn_mn_default_pipeline_v2_opt_instance.cpp PROPERTIES
+    set_source_files_properties(device_gemm_xdl_f16_f16_f16/mk_kn_mn_default_pipeline_v2_opt_instance.cpp PROPERTIES
-   COMPILE_OPTIONS "${MAX_ILP_OPTS}"
+       COMPILE_OPTIONS ";;"
-   COMPILE_DEFINITIONS "${WAVES_PER_EU_DEFS}")
+       COMPILE_DEFINITIONS "${WAVES_PER_EU_DEFS};${IGLP_OPT_DEFS}")
-# layout=TN
+    # layout=TN
-set_source_files_properties(device_gemm_xdl_f16_f16_f16/mk_nk_mn_default_pipeline_v2_opt_instance.cpp PROPERTIES
+    set_source_files_properties(device_gemm_xdl_f16_f16_f16/mk_nk_mn_default_pipeline_v2_opt_instance.cpp PROPERTIES
-   COMPILE_OPTIONS "${MAX_ILP_OPTS}"
+       COMPILE_OPTIONS "${MAX_ILP_OPTS}"
-   COMPILE_DEFINITIONS "${WAVES_PER_EU_DEFS};${IGLP_OPT_DEFS}")
+       COMPILE_DEFINITIONS "${WAVES_PER_EU_DEFS};${IGLP_OPT_DEFS}")
-endif(ENABLE_PIPELINE_V2_OPT)
+  endif(ENABLE_PIPELINE_V2_OPT)
+endif(DTYPES MATCHES "fp16" OR NOT DEFINED DTYPES)
--- a/library/src/tensor_operation_instance/gpu/gemm/device_gemm_xdl_f16_f16_f16/mk_kn_mn_default_pipeline_v2_opt_instance.cpp
+++ b/library/src/tensor_operation_instance/gpu/gemm/device_gemm_xdl_f16_f16_f16/mk_kn_mn_default_pipeline_v2_opt_instance.cpp
@@ -18,7 +18,7 @@ using Instances =
        //##########|  Type|  Type|  Type|    Type|        |        |        | Elementwise| Elementwise| Elementwise|Specialization|  Size| Block| Block| Block|   |  XDL|  XDL|  Per|  Per|   ThreadCluster|  ThreadCluster| SrcAccessOrder|   SrcVectorDim|      SrcScalar|      DstScalar| AddExtraM|   ThreadCluster|  ThreadCluster| SrcAccessOrder|  SrcVectorDim|      SrcScalar|      DstScalar| AddExtraN| SrcDstVectorDim|       DstScalar|            |                       |                      |
        //##########|      |      |      |        |        |        |        |   Operation|   Operation|   Operation|              |      |      |      |      |   |     |     | Wave| Wave| Lengths_K0_M_K1|   ArrangeOrder|               |               |      PerVector|   PerVector_K1|          | Lengths_K0_N_K1|   ArrangeOrder|               |              |      PerVector|   PerVector_K1|          |                |       PerVector|            |                       |                      |
        //##########|      |      |      |        |        |        |        |            |            |            |              |      |      |      |      |   |     |     |     |     |                |               |               |               |               |               |          |                |               |               |              |               |               |          |                |                |            |                       |                      |
-        DeviceGemmXdl<  F16,   F16,   F16,     F32,     Row,      Row,    Row, PassThrough, PassThrough, PassThrough,   GemmDefault,   256,    64,   128,     8,  8,   32,   32,    1,    2,     S<4, 64, 1>,     S<1, 0, 2>,     S<1, 0, 2>,              2,              8,              8,      true,     S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              2,              8,      true,               7,               1,           1,  LoopScheduler::Default,  PipelineVersion::v2>
+        DeviceGemmXdl<  F16,   F16,   F16,     F32,     Row,      Row,    Row, PassThrough, PassThrough, PassThrough,   GemmDefault,   256,    64,   128,     8,  8,   32,   32,    1,    2,     S<8, 32, 1>,     S<1, 0, 2>,     S<1, 0, 2>,              2,              8,              8,      true,     S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              2,              8,      true,               7,               1,           1,  LoopScheduler::Default,  PipelineVersion::v2>
 #endif
        // clang-format on
        >;

--- a/profiler/src/profile_gemm.cpp
+++ b/profiler/src/profile_gemm.cpp
@@ -121,7 +121,10 @@ int profile_gemm(int argc, char* argv[])
        return pass ? 0 : 1;
    };
-    if(data_type == GemmDataType::F32_F32_F32 && layout == GemmMatrixLayout::MK_KN_MN)
+    if(false)
+        ;
+#ifdef __fp32__
+    else if(data_type == GemmDataType::F32_F32_F32 && layout == GemmMatrixLayout::MK_KN_MN)
    {
        return profile(Row{}, Row{}, Row{}, F32{}, F32{}, F32{}, F32{});
    }
@@ -137,6 +140,8 @@ int profile_gemm(int argc, char* argv[])
    {
        return profile(Col{}, Col{}, Row{}, F32{}, F32{}, F32{}, F32{});
    }
+#endif
+#ifdef __fp16__
    else if(data_type == GemmDataType::F16_F16_F16 && layout == GemmMatrixLayout::MK_KN_MN)
    {
        return profile(Row{}, Row{}, Row{}, F16{}, F16{}, F32{}, F16{});
@@ -153,6 +158,7 @@ int profile_gemm(int argc, char* argv[])
    {
        return profile(Col{}, Col{}, Row{}, F16{}, F16{}, F32{}, F16{});
    }
+#endif
 #ifdef __bf16__
    else if(data_type == GemmDataType::BF16_BF16_BF16 && layout == GemmMatrixLayout::MK_KN_MN)
    {