Enable Async Copy for MI355 (#2425)

* add for async load builtin * add async load api * fix some compiling errors * fix a compiling error * fix some compiling errors * add a pipeline which copies from v4 * add a new pipeline for async load * fix some compiling errors * add async load tests * fix some issues in async load * fix * fix async inline assembly * fix async inline assembly * add ignore header file * comment some not gfx950 codes * comment some not gfx950 codes * fix a error * update async load apis * fix lds descriptor * fix a compiling error * fix some compiling errors * fix a descriptor issue * update lds descriptor * change async pipeline's tile distribution pattern from thread to warp * fix clang format * update async policy * fix a CRTP issue * fix a typo error * change lds layout * fix some sync issues * improve codes * delete the async test * fix a commented format issue * avoid compiling device functions when compile host * make gemm run * add the copy kernel support * finish the feature * Address comment * add the support for buffer_builtin * solved the merging problem * Comment Addressed --------- Co-authored-by: joye <joye@amd.com> Co-authored-by: joyeamd <John.Ye@amd.com> [ROCm/composable_kernel commit: f240ae3248]
2026-07-19 02:01:01 +00:00 · 2025-07-07 10:08:49 -07:00
parent 67545a9d22
commit a01042c3cf
12 changed files with 225 additions and 143 deletions
--- a/example/ck_tile/03_gemm/gemm_utils.hpp
+++ b/example/ck_tile/03_gemm/gemm_utils.hpp
@@ -15,7 +15,6 @@
 #define CK_TILE_PIPELINE_COMPUTE_V4 3
 #define CK_TILE_PIPELINE_COMPUTE_V5 4

-// temporary workaround to get k_warp_tile based on PrecType and gfx950 or not
 template <typename PrecType, ck_tile::index_t M_Warp_Tile>
 constexpr ck_tile::index_t get_k_warp_tile()
 {
--- a/example/ck_tile/36_copy/test_copy.cpp
+++ b/example/ck_tile/36_copy/test_copy.cpp
@@ -53,16 +53,17 @@ bool run(const ck_tile::ArgParser& arg_parser)

    x_buf.ToDevice(x_host.data());

-    using BlockWaves = ck_tile::sequence<2, 1>;
-    using BlockTile  = ck_tile::sequence<64, 8>;
-    using WaveTile   = ck_tile::sequence<64, 8>;
-    using Vector     = ck_tile::sequence<1, 4>;
+    using BlockWaves         = ck_tile::sequence<2, 1>;
+    using BlockTile          = ck_tile::sequence<64, 8>;
+    using WaveTile           = ck_tile::sequence<64, 8>;
+    using Vector             = ck_tile::sequence<1, 2>;
+    constexpr bool AsyncCopy = true;

    ck_tile::index_t kGridSize = (m / BlockTile::at(ck_tile::number<0>{}));
    std::cout << "grid size " << kGridSize << std::endl;

    using Shape   = ck_tile::TileCopyShape<BlockWaves, BlockTile, WaveTile, Vector>;
-    using Problem = ck_tile::TileCopyProblem<XDataType, Shape>;
+    using Problem = ck_tile::TileCopyProblem<XDataType, Shape, AsyncCopy>;
    using Kernel  = ck_tile::TileCopy<Problem>;

    constexpr ck_tile::index_t kBlockSize  = 128;
--- a/example/ck_tile/36_copy/test_copy.hpp
+++ b/example/ck_tile/36_copy/test_copy.hpp
@@ -50,11 +50,12 @@ struct TileCopyShape
    static_assert(WaveGroupSize == WarpPerBlock_M * WarpPerBlock_N, "Inconsisten wave group size!");
 };

-template <typename XDataType_, typename BlockShape_>
+template <typename XDataType_, typename BlockShape_, bool AsyncCopy_>
 struct TileCopyProblem
 {
-    using XDataType  = remove_cvref_t<XDataType_>;
-    using BlockShape = remove_cvref_t<BlockShape_>;
+    using XDataType                 = remove_cvref_t<XDataType_>;
+    using BlockShape                = remove_cvref_t<BlockShape_>;
+    static constexpr bool AsyncCopy = AsyncCopy_;
 };

 template <typename Problem_>
@@ -63,6 +64,8 @@ struct TileCopy
    using Problem   = ck_tile::remove_cvref_t<Problem_>;
    using XDataType = typename Problem::XDataType;

+    static constexpr bool AsyncCopy = Problem::AsyncCopy;
+
    template <typename Problem>
    CK_TILE_DEVICE static constexpr auto MakeDRAMDistribution()
    {
@@ -156,17 +159,29 @@ struct TileCopy

            if(my_id == warp_id)
            {
-                // load from DRAM to registers
-                load_tile(dram_tile, x_block_window);
+                if constexpr(AsyncCopy)
+                {
+                    async_load_tile(x_block_lds_window_no_dist, x_block_window);

-                // store in lds
-                store_tile(x_block_lds_window_no_dist, dram_tile);
+                    load_tile(dram_tile, x_block_lds_window);

-                // read from lds to registers
-                load_tile(dram_tile, x_block_lds_window);
+                    // store from registers to DRAM
+                    store_tile(y_block_window, dram_tile);
+                }
+                else
+                {
+                    // load from DRAM to registers
+                    load_tile(dram_tile, x_block_window);

-                // store from registers to DRAM
-                store_tile(y_block_window, dram_tile);
+                    // store in lds
+                    store_tile(x_block_lds_window_no_dist, dram_tile);
+
+                    // read from lds to registers
+                    load_tile(dram_tile, x_block_lds_window);
+
+                    // store from registers to DRAM
+                    store_tile(y_block_window, dram_tile);
+                }
            }
            __syncthreads();
            move_tile_window(x_block_window, {0, S::Block_N});