mirror of
https://github.com/ROCm/composable_kernel.git
synced 2026-04-20 14:59:17 +00:00
[rocm-libraries] ROCm/rocm-libraries#4302 (commit e62bd8a)
[CK_TILE] add tf32 support MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Proposed changes TF32 is added in CK on gfx942 and gfx950. This PR is to initiate tf32 in CK_TILE on gfx942 and gfx950. ## Checklist Please put an into the boxes that apply. You can also fill these out after creating the PR. If you're not sure, please don't hesitate to ask. - [ ] I have added tests relevant to the introduced functionality, and the unit tests are passing locally - [ ] I have added the test to REGRESSION_TESTS list defined at the top of CMakeLists.txt in tests/CMakeLists.txt, **IF** the test takes more than 30 seconds to run. - [ ] I have added inline documentation which enables the maintainers with understanding the motivation - [ ] I have removed the stale documentation which is no longer relevant after this pull request - [ ] (If this change is user-facing) I have added release notes which provide the end users with a brief summary of the improvement from this pull request - [x] I have run on all changed files - [ ] Any dependent changes have been merged ## Discussion
This commit is contained in:
committed by
assistant-librarian[bot]
parent
652d3456ca
commit
d460ab35b6
@@ -229,15 +229,6 @@ CK_TILE_DEVICE fp16x2_t cvt_pk_fp16_f32(float a, float b)
|
||||
return result;
|
||||
}
|
||||
|
||||
CK_TILE_DEVICE bf16x2_t cvt_pk_bf16_f32(float a, float b)
|
||||
{
|
||||
bf16x2_t result;
|
||||
asm volatile("v_cvt_pk_bf16_f32 %[result], %[a], %[b]"
|
||||
: [result] "=v"(result)
|
||||
: [a] "v"(a), [b] "v"(b));
|
||||
return result;
|
||||
}
|
||||
|
||||
CK_TILE_DEVICE fp32x2_t pk_mul_f32(fp32x2_t lhs, fp32x2_t rhs)
|
||||
{
|
||||
fp32x2_t result;
|
||||
@@ -856,7 +847,7 @@ struct BlockFmhaFwdV3Pipeline
|
||||
}
|
||||
else
|
||||
{
|
||||
auto casted = detail::cvt_pk_bf16_f32(x, y);
|
||||
auto casted = ck_tile::cvt_pk_bf16_f32(x, y);
|
||||
sp(sp_reg_idx).p.thread_buf_[idx] = casted.x;
|
||||
sp(sp_reg_idx).p.thread_buf_[idx + 1] = casted.y;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user