[llvm] [AMDGPU] - Add constant folding for s_bitreplicate (PR #72366)
Jessica Del via llvm-commits
llvm-commits at lists.llvm.org
Wed Nov 15 01:37:04 PST 2023
https://github.com/OutOfCache created https://github.com/llvm/llvm-project/pull/72366
If the input is a constant, we can constant fold the s_bitreplicate operation.
>From 44b4463d42b206f69114859fb04a5caeadcd3312 Mon Sep 17 00:00:00 2001
From: Jessica Del <Jessica.Del at amd.com>
Date: Tue, 14 Nov 2023 21:44:56 +0100
Subject: [PATCH] [AMDGPU] - Add constant folding for s_bitreplicate
If the input is a constant, we can constant fold the s_bitreplicate
operation.
---
llvm/lib/Analysis/ConstantFolding.cpp | 18 ++++++++++++++++++
.../CodeGen/AMDGPU/llvm.amdgcn.bitreplicate.ll | 4 ++--
2 files changed, 20 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Analysis/ConstantFolding.cpp b/llvm/lib/Analysis/ConstantFolding.cpp
index 966a65ac26b8017..e401efa6c67c2da 100644
--- a/llvm/lib/Analysis/ConstantFolding.cpp
+++ b/llvm/lib/Analysis/ConstantFolding.cpp
@@ -1533,6 +1533,7 @@ bool llvm::canConstantFoldCallTo(const CallBase *Call, const Function *F) {
case Intrinsic::amdgcn_perm:
case Intrinsic::amdgcn_wave_reduce_umin:
case Intrinsic::amdgcn_wave_reduce_umax:
+ case Intrinsic::amdgcn_s_bitreplicate:
case Intrinsic::arm_mve_vctp8:
case Intrinsic::arm_mve_vctp16:
case Intrinsic::arm_mve_vctp32:
@@ -2422,6 +2423,23 @@ static Constant *ConstantFoldScalarCall1(StringRef Name,
return ConstantFP::get(Ty->getContext(), Val);
}
+
+ case Intrinsic::amdgcn_s_bitreplicate: {
+ uint64_t Val = Op->getZExtValue();
+ uint64_t ReplicatedVal = 0;
+ uint64_t ReplicatedOnes = 0b11;
+ // Input operand is always b32
+ for (unsigned i = 0; i < 32; ++i, ReplicatedOnes <<= 2, Val >>= 1) {
+ uint64_t Bit = Val & 1;
+
+ if (!Bit)
+ continue;
+
+ ReplicatedVal |= ReplicatedOnes;
+ }
+ return ConstantInt::get(Ty, ReplicatedVal);
+ }
+
default:
return nullptr;
}
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.bitreplicate.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.bitreplicate.ll
index 027c9ef5e7cc349..0937280e6352c80 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.bitreplicate.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.bitreplicate.ll
@@ -8,8 +8,8 @@ define i64 @test_s_bitreplicate_constant() {
; GFX11-LABEL: test_s_bitreplicate_constant:
; GFX11: ; %bb.0: ; %entry
; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX11-NEXT: s_bitreplicate_b64_b32 s[0:1], 0x85fe3a92
-; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GFX11-NEXT: v_mov_b32_e32 v0, 0xfccc30c
+; GFX11-NEXT: v_mov_b32_e32 v1, 0xc033fffc
; GFX11-NEXT: s_setpc_b64 s[30:31]
entry:
%br = call i64 @llvm.amdgcn.s.bitreplicate(i32 u0x85FE3A92)
More information about the llvm-commits
mailing list