mirror of
https://github.com/ROCm/composable_kernel.git
synced 2026-05-04 13:41:24 +00:00
[CK-Tile] Add the API to load SGPR (#2878)
* Have a workable version for SGPR * have a workable version for atomic add * Revert "have a workable version for atomic add" This reverts commit 792377a590c26cfff9c8f545d9a9e8484a7422eb. * substitute with the new sgpr read api * update the CHANGELOG * have a workable version for atomic add * Revert "have a workable version for atomic add" This reverts commit 792377a590c26cfff9c8f545d9a9e8484a7422eb. * change to static for logic * have a workable version for atomic add * Revert "have a workable version for atomic add" This reverts commit 792377a590c26cfff9c8f545d9a9e8484a7422eb.
This commit is contained in:
@@ -184,17 +184,17 @@ struct FusedMoeGemmPipeline_FlatmmUk
|
||||
index_t nr_1 = kargs.hidden_size / BlockShape::Warp_N1;
|
||||
index_t kr_1 = shared_intermediate_size_1 / BlockShape::Warp_K1;
|
||||
|
||||
const IndexDataType expert_id = __builtin_amdgcn_readfirstlane(
|
||||
const IndexDataType expert_id = amd_wave_read_first_lane(
|
||||
reinterpret_cast<const IndexDataType*>(kargs.sorted_expert_ids_ptr)[sorted_tile_id]);
|
||||
index_t expert_stride_0 = shared_intermediate_size_0 * kargs.hidden_size;
|
||||
index_t expert_stride_1 = shared_intermediate_size_1 * kargs.hidden_size;
|
||||
|
||||
// nr*kr*w
|
||||
index_t interm_idx_nr0 = __builtin_amdgcn_readfirstlane(
|
||||
index_t interm_idx_nr0 = amd_wave_read_first_lane(
|
||||
intermediate_tile_id *
|
||||
BlockShape::Block_Nr0); // intermediate_tile_id * Block_N / (N in W)
|
||||
|
||||
index_t interm_idx_kr1 = __builtin_amdgcn_readfirstlane(
|
||||
index_t interm_idx_kr1 = amd_wave_read_first_lane(
|
||||
intermediate_tile_id *
|
||||
BlockShape::Block_Kr1); // intermediate_tile_id * Block_N / (N in W)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user