WMMA support for GEMM reduce (#2823)

Added gemm + reduce instance library for RDNA4. This includes: - New device implementation running GEMM and reduction kernel - instances for wmma (xdl parity) - examples for wmma (xdl parity) - tests for existing xdl and wmma [ROCm/composable_kernel commit: b25d4d684a]
2026-07-15 11:34:54 +00:00 · 2025-09-12 21:36:43 +02:00
parent 8c0cdebe63
commit f2edb06bb0
27 changed files with 1911 additions and 89 deletions
--- a/library/include/ck/library/tensor_operation_instance/gpu/gemm_universal_reduce.hpp
+++ b/library/include/ck/library/tensor_operation_instance/gpu/gemm_universal_reduce.hpp
@@ -1,5 +1,5 @@
 // SPDX-License-Identifier: MIT
-// Copyright (c) 2018-2024, Advanced Micro Devices, Inc. All rights reserved.
+// Copyright (c) 2018-2025, Advanced Micro Devices, Inc. All rights reserved.

 #pragma once

@@ -8,6 +8,7 @@
 #include "ck/ck.hpp"
 #include "ck/tensor_operation/gpu/device/tensor_layout.hpp"
 #include "ck/tensor_operation/gpu/device/impl/device_gemm_xdl_cshuffle_v3r1.hpp"
+#include "ck/tensor_operation/gpu/device/impl/device_gemm_wmma_cshuffle_v3r1.hpp"
 #include "ck/tensor_operation/gpu/element/element_wise_operation.hpp"

 #include "ck/library/tensor_operation_instance/device_operation_instance_factory.hpp"
@@ -20,6 +21,7 @@ namespace instance {
 using DsLayout   = ck::Tuple<>;
 using DsDataType = ck::Tuple<>;

+#ifdef CK_USE_XDL
 #ifdef CK_ENABLE_FP16
 void add_device_gemm_xdl_universal_reduce_f16_f16_f16_mk_kn_mn_comp_default_instances(
    std::vector<std::unique_ptr<DeviceGemmV2R1<Row,
@@ -326,7 +328,54 @@ void add_device_gemm_xdl_universal_reduce_bf16_bf16_bf16_mk_kn_mn_mem_v2_mnkpadd
                                               PassThrough,
                                               PassThrough,
                                               PassThrough>>>& instances);
+#endif
+#endif

+#ifdef CK_USE_WMMA
+#if defined(CK_ENABLE_FP16)
+void add_device_gemm_wmma_universal_reduce_f16_f16_f16_mk_kn_mn_comp_default_instances(
+    std::vector<std::unique_ptr<DeviceGemmV2R1<Row,
+                                               Row,
+                                               DsLayout,
+                                               Row,
+                                               F16,
+                                               F16,
+                                               DsDataType,
+                                               F16,
+                                               PassThrough,
+                                               PassThrough,
+                                               PassThrough>>>& instances);
+#endif
+
+#if(defined(CK_ENABLE_BF16) || defined(CK_ENABLE_INT8))
+void add_device_gemm_wmma_universal_reduce_bf16_i8_bf16_mk_kn_mn_comp_default_instances(
+    std::vector<std::unique_ptr<DeviceGemmV2R1<Row,
+                                               Row,
+                                               DsLayout,
+                                               Row,
+                                               BF16,
+                                               I8,
+                                               DsDataType,
+                                               BF16,
+                                               PassThrough,
+                                               PassThrough,
+                                               PassThrough>>>& instances);
+#endif
+
+#if defined(CK_ENABLE_BF16)
+void add_device_gemm_wmma_universal_reduce_bf16_bf16_bf16_mk_kn_mn_comp_default_instances(
+    std::vector<std::unique_ptr<DeviceGemmV2R1<Row,
+                                               Row,
+                                               DsLayout,
+                                               Row,
+                                               BF16,
+                                               BF16,
+                                               DsDataType,
+                                               BF16,
+                                               PassThrough,
+                                               PassThrough,
+                                               PassThrough>>>& instances);
+#endif
 #endif

 template <typename ADataType,
@@ -373,6 +422,7 @@ struct DeviceOperationInstanceFactory<
            if constexpr(is_same_v<ALayout, Row> && is_same_v<BLayout, Row> &&
                         is_same_v<CLayout, Row>)
            {
+#ifdef CK_USE_XDL
                add_device_gemm_xdl_universal_reduce_f16_f16_f16_mk_kn_mn_comp_default_instances(
                    op_ptrs);
                add_device_gemm_xdl_universal_reduce_f16_f16_f16_mk_kn_mn_comp_kpadding_instances(
@@ -395,6 +445,12 @@ struct DeviceOperationInstanceFactory<
                    op_ptrs);
                add_device_gemm_xdl_universal_reduce_f16_f16_f16_mk_kn_mn_mem_v2_mnkpadding_instances(
                    op_ptrs);
+#endif
+
+#ifdef CK_USE_WMMA
+                add_device_gemm_wmma_universal_reduce_f16_f16_f16_mk_kn_mn_comp_default_instances(
+                    op_ptrs);
+#endif
            }
        }
 #endif
@@ -406,6 +462,7 @@ struct DeviceOperationInstanceFactory<
            if constexpr(is_same_v<ALayout, Row> && is_same_v<BLayout, Row> &&
                         is_same_v<CLayout, Row>)
            {
+#ifdef CK_USE_XDL
                add_device_gemm_xdl_universal_reduce_bf16_i8_bf16_mk_kn_mn_comp_default_instances(
                    op_ptrs);
                add_device_gemm_xdl_universal_reduce_bf16_i8_bf16_mk_kn_mn_comp_kpadding_instances(
@@ -420,6 +477,12 @@ struct DeviceOperationInstanceFactory<
                    op_ptrs);
                add_device_gemm_xdl_universal_reduce_bf16_i8_bf16_mk_kn_mn_mem_v2_mnkpadding_instances(
                    op_ptrs);
+#endif
+
+#ifdef CK_USE_WMMA
+                add_device_gemm_wmma_universal_reduce_bf16_i8_bf16_mk_kn_mn_comp_default_instances(
+                    op_ptrs);
+#endif
            }
        }
 #endif
@@ -430,6 +493,7 @@ struct DeviceOperationInstanceFactory<
            if constexpr(is_same_v<ALayout, Row> && is_same_v<BLayout, Row> &&
                         is_same_v<CLayout, Row>)
            {
+#ifdef CK_USE_XDL
                add_device_gemm_xdl_universal_reduce_bf16_bf16_bf16_mk_kn_mn_comp_default_instances(
                    op_ptrs);
                add_device_gemm_xdl_universal_reduce_bf16_bf16_bf16_mk_kn_mn_comp_kpadding_instances(
@@ -444,6 +508,12 @@ struct DeviceOperationInstanceFactory<
                    op_ptrs);
                add_device_gemm_xdl_universal_reduce_bf16_bf16_bf16_mk_kn_mn_mem_v2_mnkpadding_instances(
                    op_ptrs);
+#endif
+
+#ifdef CK_USE_WMMA
+                add_device_gemm_wmma_universal_reduce_bf16_bf16_bf16_mk_kn_mn_comp_default_instances(
+                    op_ptrs);
+#endif
            }
        }
 #endif
--- a/library/src/tensor_operation_instance/gpu/gemm_universal_reduce/CMakeLists.txt
+++ b/library/src/tensor_operation_instance/gpu/gemm_universal_reduce/CMakeLists.txt
@@ -1,6 +1,7 @@
-# ONLY XDL_KERNELS
+# ONLY XDL_AND_WMMA_KERNELS
 set(GEMM_UNIVERSAL_REDUCE_INSTANCES)

+# XDL instances
 list(APPEND GEMM_UNIVERSAL_REDUCE_INSTANCES
        device_gemm_xdl_universal_bf16_i8_bf16/device_gemm_xdl_universal_bf16_i8_bf16_mk_kn_mn_comp_default_instance.cpp
        device_gemm_xdl_universal_bf16_i8_bf16/device_gemm_xdl_universal_bf16_i8_bf16_mk_kn_mn_comp_kpadding_instance.cpp
@@ -30,4 +31,11 @@ list(APPEND GEMM_UNIVERSAL_REDUCE_INSTANCES
        device_gemm_xdl_universal_f16_f16_f16/device_gemm_xdl_universal_f16_f16_f16_mk_kn_mn_mem_v2_mnkpadding_instance.cpp
        )

+# WMMA instances
+list(APPEND GEMM_UNIVERSAL_REDUCE_INSTANCES
+        device_gemm_wmma_universal_bf16_bf16_bf16/device_gemm_wmma_universal_bf16_bf16_bf16_mk_kn_mn_comp_default_instance.cpp
+        device_gemm_wmma_universal_bf16_i8_bf16/device_gemm_wmma_universal_bf16_i8_bf16_mk_kn_mn_comp_default_instance.cpp
+        device_gemm_wmma_universal_f16_f16_f16/device_gemm_wmma_universal_f16_f16_f16_mk_kn_mn_comp_default_instance.cpp
+        )
+
 add_instance_library(device_gemm_universal_reduce_instance ${GEMM_UNIVERSAL_REDUCE_INSTANCES})
--- a/library/src/tensor_operation_instance/gpu/gemm_universal_reduce/device_gemm_wmma_universal_bf16_bf16_bf16/device_gemm_wmma_universal_bf16_bf16_bf16_mk_kn_mn.hpp
+++ b/library/src/tensor_operation_instance/gpu/gemm_universal_reduce/device_gemm_wmma_universal_bf16_bf16_bf16/device_gemm_wmma_universal_bf16_bf16_bf16_mk_kn_mn.hpp
@@ -0,0 +1,72 @@
+// SPDX-License-Identifier: MIT
+// Copyright (c) 2025, Advanced Micro Devices, Inc. All rights reserved.
+
+#include "ck/ck.hpp"
+#include "ck/tensor_operation/gpu/device/tensor_layout.hpp"
+#include "ck/tensor_operation/gpu/device/gemm_specialization.hpp"
+#include "ck/tensor_operation/gpu/device/impl/device_gemm_wmma_cshuffle_v3r1.hpp"
+
+#include "ck/library/tensor_operation_instance/add_device_operation_instance.hpp"
+
+namespace ck {
+namespace tensor_operation {
+namespace device {
+namespace instance {
+
+using BF16 = bhalf_t;
+using F32  = float;
+
+using Row = tensor_layout::gemm::RowMajor;
+using Col = tensor_layout::gemm::ColumnMajor;
+
+template <index_t... Is>
+using S = Sequence<Is...>;
+
+using PassThrough = element_wise::PassThrough;
+
+using DsLayout   = ck::Tuple<>;
+using DsDataType = ck::Tuple<>;
+
+static constexpr auto GemmDefault    = GemmSpecialization::Default;
+static constexpr auto GemmKPadding   = GemmSpecialization::KPadding;
+static constexpr auto GemmMNPadding  = GemmSpecialization::MNPadding;
+static constexpr auto GemmMNKPadding = GemmSpecialization::MNKPadding;
+
+static constexpr auto Intrawave = BlockGemmPipelineScheduler::Intrawave;
+static constexpr auto Interwave = BlockGemmPipelineScheduler::Interwave;
+
+template <GemmSpecialization GemmSpec,
+          typename DsLayout   = ck::Tuple<>,
+          typename DsDataType = ck::Tuple<>>
+using device_gemm_wmma_universal_reduce_bf16_bf16_bf16_mk_kn_mn_instances =
+    std::tuple<
+        // clang-format off
+        //#########################| ALayout| BLayout|  DsLayout| CLayout| AData| BData|      DsData| CData| AccData| Cshuffle|           A|           B|           C|          GEMM| Block|  MPer|  NPer|  KPer| AK1| BK1|MPerWmma|NPerWmma|MRepeat|NRepeat|  ABlockTransfer| ABlockTransfer| ABlockTransfer| ABlockTransfer| ABlockTransfer| ABlockTransfer| ABlockLds|  BBlockTransfer| BBlockTransfer| BBlockTransfer| BlockTransfer| BBlockTransfer| BBlockTransfer| BBlockLds|  CShuffle|  CShuffle|         CBlockTransferClusterLengths|  CBlockTransfer|                         Block-wiseGemm|                Block-wiseGemm|      Reduce|
+        //#########################|        |        |          |        |  Type|  Type|        Type|  Type|    Type|     Type| Elementwise| Elementwise| Elementwise|Specialization|  Size| Block| Block| Block|    |    |        |        |       |       |   ThreadCluster|  ThreadCluster| SrcAccessOrder|   SrcVectorDim|      SrcScalar|      DstScalar| AddExtraM|   ThreadCluster|  ThreadCluster| SrcAccessOrder|  SrcVectorDim|      SrcScalar|      DstScalar| AddExtraN|MRepeatPer|NRepeatPer|     _MBlock_MRepeatPerShuffle_MWaveM| ScalarPerVector|                               Pipeline|                      Pipeline|    DataType|
+        //#########################|        |        |          |        |      |      |            |      |        |         |   Operation|   Operation|   Operation|              |      |      |      |      |    |    |        |        |       |       | Lengths_K0_M_K1|   ArrangeOrder|               |               |      PerVector|   PerVector_K1|          | Lengths_K0_N_K1|   ArrangeOrder|               |              |      PerVector|   PerVector_K1|          | Shuffle  | Shuffle  |  PerShuffle_NBlock_NRepeatPerShuffle|   _NPerBlock   |                              Scheduler|                       Version|            |
+        //#########################|        |        |          |        |      |      |            |      |        |         |            |            |            |              |      |      |      |      |    |    |        |        |       |       |                |               |               |               |               |               |          |                |               |               |              |               |               |          |          |          | _NWaveNPerRepeat                    |                |                                       |                              |            |
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    32,   8,   8,      16,      16,      4,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    32,   8,   8,      16,      16,      4,      4,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   256,   128,    32,   8,   8,      16,      16,      8,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    64,   8,   8,      16,      16,      4,      2,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    64,   8,   8,      16,      16,      4,      4,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,    64,    64,    32,   8,   8,      16,      16,      2,      2,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              2,              4,     true,          1,         1,                       S<1, 32, 1, 2>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,   128,    64,    64,   8,   8,      16,      16,      4,      2,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 4>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    32,   8,   8,      16,      16,      4,      4,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Interwave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    64,   8,   8,      16,      16,      4,      2,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Interwave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,   128,   128,    32,   8,   8,      16,      16,      4,      4,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 4>,               8,  BlockGemmPipelineScheduler::Interwave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    32,   8,   8,      16,      16,      4,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Interwave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    32,   8,   8,      16,      16,      4,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    32,   8,   8,      16,      16,      4,      4,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   256,   128,    32,   8,   8,      16,      16,      8,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    64,   8,   8,      16,      16,      4,      2,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    64,   8,   8,      16,      16,      4,      4,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,    64,    64,    32,   8,   8,      16,      16,      2,      2,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              2,              4,     true,          1,         1,                       S<1, 32, 1, 2>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,   BF16, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,   128,    64,    64,   8,   8,      16,      16,      4,      2,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 4>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>
+        // clang-format on
+        >;
+
+} // namespace instance
+} // namespace device
+} // namespace tensor_operation
+} // namespace ck
--- a/library/src/tensor_operation_instance/gpu/gemm_universal_reduce/device_gemm_wmma_universal_bf16_bf16_bf16/device_gemm_wmma_universal_bf16_bf16_bf16_mk_kn_mn_comp_default_instance.cpp
+++ b/library/src/tensor_operation_instance/gpu/gemm_universal_reduce/device_gemm_wmma_universal_bf16_bf16_bf16/device_gemm_wmma_universal_bf16_bf16_bf16_mk_kn_mn_comp_default_instance.cpp
@@ -0,0 +1,58 @@
+// SPDX-License-Identifier: MIT
+// Copyright (c) 2025, Advanced Micro Devices, Inc. All rights reserved.
+
+#include "device_gemm_wmma_universal_bf16_bf16_bf16_mk_kn_mn.hpp"
+#include "ck/host_utility/device_prop.hpp"
+#include "ck/tensor_operation/gpu/element/element_wise_operation.hpp"
+
+namespace ck {
+namespace tensor_operation {
+namespace device {
+namespace instance {
+
+using BF16        = bhalf_t;
+using Row         = tensor_layout::gemm::RowMajor;
+using PassThrough = element_wise::PassThrough;
+
+void add_device_gemm_wmma_universal_reduce_bf16_bf16_bf16_mk_kn_mn_comp_default_instances(
+    std::vector<std::unique_ptr<DeviceGemmV2R1<Row,
+                                               Row,
+                                               DsLayout,
+                                               Row,
+                                               BF16,
+                                               BF16,
+                                               DsDataType,
+                                               BF16,
+                                               PassThrough,
+                                               PassThrough,
+                                               PassThrough>>>& instances)
+{
+    if(ck::is_gfx12_supported())
+    {
+        add_device_operation_instances(
+            instances,
+            device_gemm_wmma_universal_reduce_bf16_bf16_bf16_mk_kn_mn_instances<GemmDefault,
+                                                                                DsLayout,
+                                                                                DsDataType>{});
+        add_device_operation_instances(
+            instances,
+            device_gemm_wmma_universal_reduce_bf16_bf16_bf16_mk_kn_mn_instances<GemmKPadding,
+                                                                                DsLayout,
+                                                                                DsDataType>{});
+        add_device_operation_instances(
+            instances,
+            device_gemm_wmma_universal_reduce_bf16_bf16_bf16_mk_kn_mn_instances<GemmMNPadding,
+                                                                                DsLayout,
+                                                                                DsDataType>{});
+        add_device_operation_instances(
+            instances,
+            device_gemm_wmma_universal_reduce_bf16_bf16_bf16_mk_kn_mn_instances<GemmMNKPadding,
+                                                                                DsLayout,
+                                                                                DsDataType>{});
+    }
+}
+
+} // namespace instance
+} // namespace device
+} // namespace tensor_operation
+} // namespace ck
--- a/library/src/tensor_operation_instance/gpu/gemm_universal_reduce/device_gemm_wmma_universal_bf16_i8_bf16/device_gemm_wmma_universal_bf16_i8_bf16_mk_kn_mn.hpp
+++ b/library/src/tensor_operation_instance/gpu/gemm_universal_reduce/device_gemm_wmma_universal_bf16_i8_bf16/device_gemm_wmma_universal_bf16_i8_bf16_mk_kn_mn.hpp
@@ -0,0 +1,73 @@
+// SPDX-License-Identifier: MIT
+// Copyright (c) 2025, Advanced Micro Devices, Inc. All rights reserved.
+
+#include "ck/ck.hpp"
+#include "ck/tensor_operation/gpu/device/tensor_layout.hpp"
+#include "ck/tensor_operation/gpu/device/gemm_specialization.hpp"
+#include "ck/tensor_operation/gpu/device/impl/device_gemm_wmma_cshuffle_v3r1.hpp"
+
+#include "ck/library/tensor_operation_instance/add_device_operation_instance.hpp"
+
+namespace ck {
+namespace tensor_operation {
+namespace device {
+namespace instance {
+
+using I8   = int8_t;
+using BF16 = bhalf_t;
+using F32  = float;
+
+using Row = tensor_layout::gemm::RowMajor;
+using Col = tensor_layout::gemm::ColumnMajor;
+
+template <index_t... Is>
+using S = Sequence<Is...>;
+
+using PassThrough = element_wise::PassThrough;
+
+using DsLayout   = ck::Tuple<>;
+using DsDataType = ck::Tuple<>;
+
+static constexpr auto GemmDefault    = GemmSpecialization::Default;
+static constexpr auto GemmKPadding   = GemmSpecialization::KPadding;
+static constexpr auto GemmMNPadding  = GemmSpecialization::MNPadding;
+static constexpr auto GemmMNKPadding = GemmSpecialization::MNKPadding;
+
+static constexpr auto Intrawave = BlockGemmPipelineScheduler::Intrawave;
+static constexpr auto Interwave = BlockGemmPipelineScheduler::Interwave;
+
+template <GemmSpecialization GemmSpec,
+          typename DsLayout   = ck::Tuple<>,
+          typename DsDataType = ck::Tuple<>>
+using device_gemm_wmma_universal_reduce_bf16_i8_bf16_mk_kn_mn_instances =
+    std::tuple<
+        // clang-format off
+        //#########################| ALayout| BLayout|  DsLayout| CLayout| AData| BData|      DsData| CData| AccData| Cshuffle|           A|           B|           C|          GEMM| Block|  MPer|  NPer|  KPer| AK1| BK1|MPerWmma|NPerWmma|MRepeat|NRepeat|  ABlockTransfer| ABlockTransfer| ABlockTransfer| ABlockTransfer| ABlockTransfer| ABlockTransfer| ABlockLds|  BBlockTransfer| BBlockTransfer| BBlockTransfer| BlockTransfer| BBlockTransfer| BBlockTransfer| BBlockLds|  CShuffle|  CShuffle|         CBlockTransferClusterLengths|  CBlockTransfer|                         Block-wiseGemm|               Block-wiseGemm|      Reduce|
+        //#########################|        |        |          |        |  Type|  Type|        Type|  Type|    Type|     Type| Elementwise| Elementwise| Elementwise|Specialization|  Size| Block| Block| Block|    |    |        |        |       |       |   ThreadCluster|  ThreadCluster| SrcAccessOrder|   SrcVectorDim|      SrcScalar|      DstScalar| AddExtraM|   ThreadCluster|  ThreadCluster| SrcAccessOrder|  SrcVectorDim|      SrcScalar|      DstScalar| AddExtraN|MRepeatPer|NRepeatPer|     _MBlock_MRepeatPerShuffle_MWaveM| ScalarPerVector|                               Pipeline|                     Pipeline|    DataType|
+        //#########################|        |        |          |        |      |      |            |      |        |         |   Operation|   Operation|   Operation|              |      |      |      |      |    |    |        |        |       |       | Lengths_K0_M_K1|   ArrangeOrder|               |               |      PerVector|   PerVector_K1|          | Lengths_K0_N_K1|   ArrangeOrder|               |              |      PerVector|   PerVector_K1|          | Shuffle  | Shuffle  |  PerShuffle_NBlock_NRepeatPerShuffle|   _NPerBlock   |                              Scheduler|                      Version|            |
+        //#########################|        |        |          |        |      |      |            |      |        |         |            |            |            |              |      |      |      |      |    |    |        |        |       |       |                |               |               |               |               |               |          |                |               |               |              |               |               |          |          |          | _NWaveNPerRepeat                    |                |                                       |                             |            |
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    32,   8,   4,      16,      16,      4,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    32,   8,   4,      16,      16,      4,      4,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   256,   128,    32,   8,   4,      16,      16,      8,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    64,   8,   4,      16,      16,      4,      2,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    64,   8,   4,      16,      16,      4,      4,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,    64,    64,    32,   8,   4,      16,      16,      2,      2,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              2,              2,     true,          1,         1,                       S<1, 32, 1, 2>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,   128,    64,    64,   8,   4,      16,      16,      4,      2,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 4>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    32,   8,   4,      16,      16,      4,      4,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Interwave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    64,   8,   4,      16,      16,      4,      2,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Interwave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,   128,   128,    32,   8,   4,      16,      16,      4,      4,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 4>,               8,  BlockGemmPipelineScheduler::Interwave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    32,   8,   4,      16,      16,      4,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Interwave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    32,   8,   4,      16,      16,      4,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    32,   8,   4,      16,      16,      4,      4,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   256,   128,    32,   8,   4,      16,      16,      8,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    64,   8,   4,      16,      16,      4,      2,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    64,   8,   4,      16,      16,      4,      4,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,    64,    64,    32,   8,   4,      16,      16,      2,      2,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              2,              2,     true,          1,         1,                       S<1, 32, 1, 2>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,  BF16,    I8, DsDataType,  BF16,     F32,     BF16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,   128,    64,    64,   8,   4,      16,      16,      4,      2,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              4,     true,          1,         1,                       S<1, 32, 1, 4>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>
+        // clang-format on
+        >;
+
+} // namespace instance
+} // namespace device
+} // namespace tensor_operation
+} // namespace ck
--- a/library/src/tensor_operation_instance/gpu/gemm_universal_reduce/device_gemm_wmma_universal_bf16_i8_bf16/device_gemm_wmma_universal_bf16_i8_bf16_mk_kn_mn_comp_default_instance.cpp
+++ b/library/src/tensor_operation_instance/gpu/gemm_universal_reduce/device_gemm_wmma_universal_bf16_i8_bf16/device_gemm_wmma_universal_bf16_i8_bf16_mk_kn_mn_comp_default_instance.cpp
@@ -0,0 +1,59 @@
+// SPDX-License-Identifier: MIT
+// Copyright (c) 2025, Advanced Micro Devices, Inc. All rights reserved.
+
+#include "device_gemm_wmma_universal_bf16_i8_bf16_mk_kn_mn.hpp"
+#include "ck/host_utility/device_prop.hpp"
+#include "ck/tensor_operation/gpu/element/element_wise_operation.hpp"
+
+namespace ck {
+namespace tensor_operation {
+namespace device {
+namespace instance {
+
+using I8          = int8_t;
+using BF16        = bhalf_t;
+using Row         = tensor_layout::gemm::RowMajor;
+using PassThrough = element_wise::PassThrough;
+
+void add_device_gemm_wmma_universal_reduce_bf16_i8_bf16_mk_kn_mn_comp_default_instances(
+    std::vector<std::unique_ptr<DeviceGemmV2R1<Row,
+                                               Row,
+                                               DsLayout,
+                                               Row,
+                                               BF16,
+                                               I8,
+                                               DsDataType,
+                                               BF16,
+                                               PassThrough,
+                                               PassThrough,
+                                               PassThrough>>>& instances)
+{
+    if(ck::is_gfx12_supported())
+    {
+        add_device_operation_instances(
+            instances,
+            device_gemm_wmma_universal_reduce_bf16_i8_bf16_mk_kn_mn_instances<GemmDefault,
+                                                                              DsLayout,
+                                                                              DsDataType>{});
+        add_device_operation_instances(
+            instances,
+            device_gemm_wmma_universal_reduce_bf16_i8_bf16_mk_kn_mn_instances<GemmKPadding,
+                                                                              DsLayout,
+                                                                              DsDataType>{});
+        add_device_operation_instances(
+            instances,
+            device_gemm_wmma_universal_reduce_bf16_i8_bf16_mk_kn_mn_instances<GemmMNPadding,
+                                                                              DsLayout,
+                                                                              DsDataType>{});
+        add_device_operation_instances(
+            instances,
+            device_gemm_wmma_universal_reduce_bf16_i8_bf16_mk_kn_mn_instances<GemmMNKPadding,
+                                                                              DsLayout,
+                                                                              DsDataType>{});
+    }
+}
+
+} // namespace instance
+} // namespace device
+} // namespace tensor_operation
+} // namespace ck
--- a/library/src/tensor_operation_instance/gpu/gemm_universal_reduce/device_gemm_wmma_universal_f16_f16_f16/device_gemm_wmma_universal_f16_f16_f16_mk_kn_mn.hpp
+++ b/library/src/tensor_operation_instance/gpu/gemm_universal_reduce/device_gemm_wmma_universal_f16_f16_f16/device_gemm_wmma_universal_f16_f16_f16_mk_kn_mn.hpp
@@ -0,0 +1,72 @@
+// SPDX-License-Identifier: MIT
+// Copyright (c) 2025, Advanced Micro Devices, Inc. All rights reserved.
+
+#include "ck/ck.hpp"
+#include "ck/tensor_operation/gpu/device/tensor_layout.hpp"
+#include "ck/tensor_operation/gpu/device/gemm_specialization.hpp"
+#include "ck/tensor_operation/gpu/device/impl/device_gemm_wmma_cshuffle_v3r1.hpp"
+
+#include "ck/library/tensor_operation_instance/add_device_operation_instance.hpp"
+
+namespace ck {
+namespace tensor_operation {
+namespace device {
+namespace instance {
+
+using F16 = half_t;
+using F32 = float;
+
+using Row = tensor_layout::gemm::RowMajor;
+using Col = tensor_layout::gemm::ColumnMajor;
+
+template <index_t... Is>
+using S = Sequence<Is...>;
+
+using PassThrough = element_wise::PassThrough;
+
+using DsLayout   = ck::Tuple<>;
+using DsDataType = ck::Tuple<>;
+
+static constexpr auto GemmDefault    = GemmSpecialization::Default;
+static constexpr auto GemmKPadding   = GemmSpecialization::KPadding;
+static constexpr auto GemmMNPadding  = GemmSpecialization::MNPadding;
+static constexpr auto GemmMNKPadding = GemmSpecialization::MNKPadding;
+
+static constexpr auto Intrawave = BlockGemmPipelineScheduler::Intrawave;
+static constexpr auto Interwave = BlockGemmPipelineScheduler::Interwave;
+
+template <GemmSpecialization GemmSpec,
+          typename DsLayout   = ck::Tuple<>,
+          typename DsDataType = ck::Tuple<>>
+using device_gemm_wmma_universal_reduce_f16_f16_f16_mk_kn_mn_instances =
+    std::tuple<
+        // clang-format off
+        //#########################| ALayout| BLayout|  DsLayout| CLayout| AData| BData|      DsData| CData| AccData| Cshuffle|           A|           B|           C|          GEMM| Block|  MPer|  NPer|  KPer| AK1| BK1|MPerWmma|NPerWmma|MRepeat|NRepeat|  ABlockTransfer| ABlockTransfer| ABlockTransfer| ABlockTransfer| ABlockTransfer| ABlockTransfer| ABlockLds|  BBlockTransfer| BBlockTransfer| BBlockTransfer| BlockTransfer| BBlockTransfer| BBlockTransfer| BBlockLds|  CShuffle|  CShuffle|         CBlockTransferClusterLengths|  CBlockTransfer|                         Block-wiseGemm|                Block-wiseGemm|      Reduce|
+        //#########################|        |        |          |        |  Type|  Type|        Type|  Type|    Type|     Type| Elementwise| Elementwise| Elementwise|Specialization|  Size| Block| Block| Block|    |    |        |        |       |       |   ThreadCluster|  ThreadCluster| SrcAccessOrder|   SrcVectorDim|      SrcScalar|      DstScalar| AddExtraM|   ThreadCluster|  ThreadCluster| SrcAccessOrder|  SrcVectorDim|      SrcScalar|      DstScalar| AddExtraN|MRepeatPer|NRepeatPer|     _MBlock_MRepeatPerShuffle_MWaveM| ScalarPerVector|                               Pipeline|                      Pipeline|    DataType|
+        //#########################|        |        |          |        |      |      |            |      |        |         |   Operation|   Operation|   Operation|              |      |      |      |      |    |    |        |        |       |       | Lengths_K0_M_K1|   ArrangeOrder|               |               |      PerVector|   PerVector_K1|          | Lengths_K0_N_K1|   ArrangeOrder|               |              |      PerVector|   PerVector_K1|          | Shuffle  | Shuffle  |  PerShuffle_NBlock_NRepeatPerShuffle|   _NPerBlock   |                              Scheduler|                       Version|            |
+        //#########################|        |        |          |        |      |      |            |      |        |         |            |            |            |              |      |      |      |      |    |    |        |        |       |       |                |               |               |               |               |               |          |                |               |               |              |               |               |          |          |          | _NWaveNPerRepeat                    |                |                                       |                              |            |
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    32,   8,   8,      16,      16,      4,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    32,   8,   8,      16,      16,      4,      4,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   256,   128,    32,   8,   8,      16,      16,      8,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    64,   8,   8,      16,      16,      4,      2,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    64,   8,   8,      16,      16,      4,      4,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,    64,    64,    32,   8,   8,      16,      16,      2,      2,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              2,              4,     true,          1,         1,                       S<1, 32, 1, 2>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,   128,    64,    64,   8,   8,      16,      16,      4,      2,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 4>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    32,   8,   8,      16,      16,      4,      4,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Interwave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    64,   8,   8,      16,      16,      4,      2,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Interwave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,   128,   128,    32,   8,   8,      16,      16,      4,      4,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 4>,               8,  BlockGemmPipelineScheduler::Interwave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    32,   8,   8,      16,      16,      4,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Interwave,  BlockGemmPipelineVersion::v1,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    32,   8,   8,      16,      16,      4,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    32,   8,   8,      16,      16,      4,      4,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   256,   128,    32,   8,   8,      16,      16,      8,      2,     S<4, 64, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 64, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   128,    64,   8,   8,      16,      16,      4,      2,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   256,   128,   256,    64,   8,   8,      16,      16,      4,      4,     S<8, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<8, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 8>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,    64,    64,    32,   8,   8,      16,      16,      2,      2,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              2,              4,     true,          1,         1,                       S<1, 32, 1, 2>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>,
+        DeviceGemm_Wmma_CShuffleV3R1<    Row,     Row,  DsLayout,     Row,   F16,   F16,  DsDataType,   F16,     F32,      F16, PassThrough, PassThrough, PassThrough,      GemmSpec,   128,   128,    64,    64,   8,   8,      16,      16,      4,      2,     S<4, 32, 1>,     S<0, 2, 1>,    S<0, 2, 1>,               1,              1,              8,       true,    S<4, 32, 1>,     S<0, 2, 1>,     S<0, 2, 1>,             1,              1,              8,     true,          1,         1,                       S<1, 32, 1, 4>,               8,  BlockGemmPipelineScheduler::Intrawave,  BlockGemmPipelineVersion::v3,      float>
+        // clang-format on
+        >;
+
+} // namespace instance
+} // namespace device
+} // namespace tensor_operation
+} // namespace ck
--- a/library/src/tensor_operation_instance/gpu/gemm_universal_reduce/device_gemm_wmma_universal_f16_f16_f16/device_gemm_wmma_universal_f16_f16_f16_mk_kn_mn_comp_default_instance.cpp
+++ b/library/src/tensor_operation_instance/gpu/gemm_universal_reduce/device_gemm_wmma_universal_f16_f16_f16/device_gemm_wmma_universal_f16_f16_f16_mk_kn_mn_comp_default_instance.cpp
@@ -0,0 +1,57 @@
+// SPDX-License-Identifier: MIT
+// Copyright (c) 2025, Advanced Micro Devices, Inc. All rights reserved.
+
+#include "device_gemm_wmma_universal_f16_f16_f16_mk_kn_mn.hpp"
+#include "ck/host_utility/device_prop.hpp"
+
+namespace ck {
+namespace tensor_operation {
+namespace device {
+namespace instance {
+
+using F16         = half_t;
+using Row         = tensor_layout::gemm::RowMajor;
+using PassThrough = element_wise::PassThrough;
+using Add         = element_wise::Add;
+
+using DsLayout_F16   = ck::Tuple<>;
+using DsDataType_F16 = ck::Tuple<>;
+
+void add_device_gemm_wmma_universal_reduce_f16_f16_f16_mk_kn_mn_comp_default_instances(
+    std::vector<std::unique_ptr<DeviceGemmV2R1<Row,
+                                               Row,
+                                               DsLayout_F16,
+                                               Row,
+                                               F16,
+                                               F16,
+                                               DsDataType_F16,
+                                               F16,
+                                               PassThrough,
+                                               PassThrough,
+                                               PassThrough>>>& instances)
+{
+    if(ck::is_gfx12_supported())
+    {
+        add_device_operation_instances(
+            instances,
+            device_gemm_wmma_universal_reduce_f16_f16_f16_mk_kn_mn_instances<GemmDefault,
+                                                                             DsLayout_F16,
+                                                                             DsDataType_F16>{});
+        add_device_operation_instances(
+            instances,
+            device_gemm_wmma_universal_reduce_f16_f16_f16_mk_kn_mn_instances<GemmKPadding,
+                                                                             DsLayout_F16,
+                                                                             DsDataType_F16>{});
+        add_device_operation_instances(
+            instances,
+            device_gemm_wmma_universal_reduce_f16_f16_f16_mk_kn_mn_instances<GemmMNPadding>{});
+        add_device_operation_instances(
+            instances,
+            device_gemm_wmma_universal_reduce_f16_f16_f16_mk_kn_mn_instances<GemmMNKPadding>{});
+    }
+}
+
+} // namespace instance
+} // namespace device
+} // namespace tensor_operation
+} // namespace ck