[llvm] 6d7e7fb - [InstCombine] Fold uitofp of masked truncations (#222592)
via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 13 17:58:54 PDT 2026
Author: Harrison Hao
Date: 2026-09-14T08:58:49+08:00
New Revision: 6d7e7fbafd8d010a3a4146b24e8cca2c8b7dc831
URL: https://github.com/llvm/llvm-project/commit/6d7e7fbafd8d010a3a4146b24e8cca2c8b7dc831
DIFF: https://github.com/llvm/llvm-project/commit/6d7e7fbafd8d010a3a4146b24e8cca2c8b7dc831.diff
LOG: [InstCombine] Fold uitofp of masked truncations (#222592)
Fold an unsigned integer-to-floating-point conversion of a masked
truncation by widening the mask and eliminating the truncation:
`uitofp((trunc X) & C)` → `uitofp(X & zext(C))`
Apply the fold when the intermediate instructions have one use and
widening the integer operation is desirable.
This reduces the instruction count on both AMDGPU and NVPTX.
AMDGPU example: https://godbolt.org/z/rTYhM61ce
NVPTX example: https://godbolt.org/z/fWhsrne19
Added:
llvm/test/Transforms/InstCombine/uitofp.ll
Modified:
llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
index 00db99611d0c2..97defc5e3ddd3 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
@@ -2630,6 +2630,26 @@ Instruction *InstCombinerImpl::visitUIToFP(CastInst &CI) {
CI.setNonNeg();
return &CI;
}
+
+ // uitofp (and (trunc X), Mask) --> uitofp (and X, zext(Mask))
+ Value *Src = CI.getOperand(0);
+ Value *X;
+ Constant *Mask;
+ if (match(Src, m_OneUse(m_And(m_OneUse(m_Trunc(m_Value(X))),
+ m_ImmConstant(Mask))))) {
+ unsigned SourceWidth = Src->getType()->getScalarSizeInBits();
+ unsigned InputWidth = X->getType()->getScalarSizeInBits();
+ if (!DL.isLegalInteger(SourceWidth) &&
+ shouldChangeType(SourceWidth, InputWidth)) {
+ Value *MaskedX =
+ Builder.CreateAnd(X, Builder.CreateZExt(Mask, X->getType()));
+ auto *NewUIToFP =
+ CastInst::Create(Instruction::UIToFP, MaskedX, CI.getType());
+ NewUIToFP->setNonNeg(CI.hasNonNeg());
+ return NewUIToFP;
+ }
+ }
+
return nullptr;
}
diff --git a/llvm/test/Transforms/InstCombine/uitofp.ll b/llvm/test/Transforms/InstCombine/uitofp.ll
new file mode 100644
index 0000000000000..ba443ae581639
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/uitofp.ll
@@ -0,0 +1,129 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt < %s -passes=instcombine -S | FileCheck %s
+
+target datalayout = "n32:64"
+
+ at external_mask = external global i8
+
+define half @uitofp_trunc_and_mask_scalar(i32 %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_scalar(
+; CHECK-NEXT: [[TMP1:%.*]] = and i32 [[X:%.*]], 255
+; CHECK-NEXT: [[RESULT:%.*]] = uitofp nneg i32 [[TMP1]] to half
+; CHECK-NEXT: ret half [[RESULT]]
+;
+ %trunc = trunc i32 %x to i16
+ %masked = and i16 %trunc, 255
+ %result = uitofp nneg i16 %masked to half
+ ret half %result
+}
+
+define <2 x half> @uitofp_trunc_and_mask_vector(<2 x i32> %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_vector(
+; CHECK-NEXT: [[TMP1:%.*]] = and <2 x i32> [[X:%.*]], splat (i32 255)
+; CHECK-NEXT: [[RESULT:%.*]] = uitofp nneg <2 x i32> [[TMP1]] to <2 x half>
+; CHECK-NEXT: ret <2 x half> [[RESULT]]
+;
+ %trunc = trunc <2 x i32> %x to <2 x i16>
+ %masked = and <2 x i16> %trunc, splat (i16 255)
+ %result = uitofp nneg <2 x i16> %masked to <2 x half>
+ ret <2 x half> %result
+}
+
+define <2 x half> @uitofp_trunc_and_constexpr_mask(<2 x i32> %x) {
+; CHECK-LABEL: @uitofp_trunc_and_constexpr_mask(
+; CHECK-NEXT: [[TRUNC:%.*]] = trunc <2 x i32> [[X:%.*]] to <2 x i16>
+; CHECK-NEXT: [[MASKED:%.*]] = and <2 x i16> [[TRUNC]], <i16 ptrtoint (ptr @external_mask to i16), i16 ptrtoint (ptr @external_mask to i16)>
+; CHECK-NEXT: [[RESULT:%.*]] = uitofp nneg <2 x i16> [[MASKED]] to <2 x half>
+; CHECK-NEXT: ret <2 x half> [[RESULT]]
+;
+ %trunc = trunc <2 x i32> %x to <2 x i16>
+ %masked = and <2 x i16> %trunc, splat (i16 ptrtoint (ptr @external_mask to i16))
+ %result = uitofp nneg <2 x i16> %masked to <2 x half>
+ ret <2 x half> %result
+}
+
+define { <2 x half>, <2 x i16> } @uitofp_trunc_and_mask_multiuse_and(<2 x i32> %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_multiuse_and(
+; CHECK-NEXT: [[TRUNC:%.*]] = trunc <2 x i32> [[X:%.*]] to <2 x i16>
+; CHECK-NEXT: [[MASKED:%.*]] = and <2 x i16> [[TRUNC]], splat (i16 255)
+; CHECK-NEXT: [[RESULT:%.*]] = uitofp nneg <2 x i16> [[MASKED]] to <2 x half>
+; CHECK-NEXT: [[RET0:%.*]] = insertvalue { <2 x half>, <2 x i16> } poison, <2 x half> [[RESULT]], 0
+; CHECK-NEXT: [[RET1:%.*]] = insertvalue { <2 x half>, <2 x i16> } [[RET0]], <2 x i16> [[MASKED]], 1
+; CHECK-NEXT: ret { <2 x half>, <2 x i16> } [[RET1]]
+;
+ %trunc = trunc <2 x i32> %x to <2 x i16>
+ %masked = and <2 x i16> %trunc, splat (i16 255)
+ %result = uitofp nneg <2 x i16> %masked to <2 x half>
+ %ret0 = insertvalue { <2 x half>, <2 x i16> } poison, <2 x half> %result, 0
+ %ret1 = insertvalue { <2 x half>, <2 x i16> } %ret0, <2 x i16> %masked, 1
+ ret { <2 x half>, <2 x i16> } %ret1
+}
+
+define { <2 x half>, <2 x i16> } @uitofp_trunc_and_mask_multiuse_trunc(<2 x i32> %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_multiuse_trunc(
+; CHECK-NEXT: [[TRUNC:%.*]] = trunc <2 x i32> [[X:%.*]] to <2 x i16>
+; CHECK-NEXT: [[MASKED:%.*]] = and <2 x i16> [[TRUNC]], splat (i16 255)
+; CHECK-NEXT: [[RESULT:%.*]] = uitofp nneg <2 x i16> [[MASKED]] to <2 x half>
+; CHECK-NEXT: [[RET0:%.*]] = insertvalue { <2 x half>, <2 x i16> } poison, <2 x half> [[RESULT]], 0
+; CHECK-NEXT: [[RET1:%.*]] = insertvalue { <2 x half>, <2 x i16> } [[RET0]], <2 x i16> [[TRUNC]], 1
+; CHECK-NEXT: ret { <2 x half>, <2 x i16> } [[RET1]]
+;
+ %trunc = trunc <2 x i32> %x to <2 x i16>
+ %masked = and <2 x i16> %trunc, splat (i16 255)
+ %result = uitofp nneg <2 x i16> %masked to <2 x half>
+ %ret0 = insertvalue { <2 x half>, <2 x i16> } poison, <2 x half> %result, 0
+ %ret1 = insertvalue { <2 x half>, <2 x i16> } %ret0, <2 x i16> %trunc, 1
+ ret { <2 x half>, <2 x i16> } %ret1
+}
+
+define float @uitofp_trunc_and_mask_legal_narrow_scalar(i64 %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_legal_narrow_scalar(
+; CHECK-NEXT: [[TRUNC:%.*]] = trunc i64 [[X:%.*]] to i32
+; CHECK-NEXT: [[MASKED:%.*]] = and i32 [[TRUNC]], 255
+; CHECK-NEXT: [[RESULT:%.*]] = uitofp nneg i32 [[MASKED]] to float
+; CHECK-NEXT: ret float [[RESULT]]
+;
+ %trunc = trunc i64 %x to i32
+ %masked = and i32 %trunc, 255
+ %result = uitofp nneg i32 %masked to float
+ ret float %result
+}
+
+define <2 x float> @uitofp_trunc_and_mask_legal_narrow_vector(<2 x i64> %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_legal_narrow_vector(
+; CHECK-NEXT: [[TRUNC:%.*]] = trunc <2 x i64> [[X:%.*]] to <2 x i32>
+; CHECK-NEXT: [[MASKED:%.*]] = and <2 x i32> [[TRUNC]], splat (i32 255)
+; CHECK-NEXT: [[RESULT:%.*]] = uitofp nneg <2 x i32> [[MASKED]] to <2 x float>
+; CHECK-NEXT: ret <2 x float> [[RESULT]]
+;
+ %trunc = trunc <2 x i64> %x to <2 x i32>
+ %masked = and <2 x i32> %trunc, splat (i32 255)
+ %result = uitofp nneg <2 x i32> %masked to <2 x float>
+ ret <2 x float> %result
+}
+
+define double @uitofp_trunc_and_mask_illegal_wide_scalar(i128 %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_illegal_wide_scalar(
+; CHECK-NEXT: [[TRUNC:%.*]] = trunc i128 [[X:%.*]] to i64
+; CHECK-NEXT: [[MASKED:%.*]] = and i64 [[TRUNC]], 255
+; CHECK-NEXT: [[RESULT:%.*]] = uitofp nneg i64 [[MASKED]] to double
+; CHECK-NEXT: ret double [[RESULT]]
+;
+ %trunc = trunc i128 %x to i64
+ %masked = and i64 %trunc, 255
+ %result = uitofp nneg i64 %masked to double
+ ret double %result
+}
+
+define <2 x double> @uitofp_trunc_and_mask_illegal_wide_vector(<2 x i128> %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_illegal_wide_vector(
+; CHECK-NEXT: [[TRUNC:%.*]] = trunc <2 x i128> [[X:%.*]] to <2 x i64>
+; CHECK-NEXT: [[MASKED:%.*]] = and <2 x i64> [[TRUNC]], splat (i64 255)
+; CHECK-NEXT: [[RESULT:%.*]] = uitofp nneg <2 x i64> [[MASKED]] to <2 x double>
+; CHECK-NEXT: ret <2 x double> [[RESULT]]
+;
+ %trunc = trunc <2 x i128> %x to <2 x i64>
+ %masked = and <2 x i64> %trunc, splat (i64 255)
+ %result = uitofp nneg <2 x i64> %masked to <2 x double>
+ ret <2 x double> %result
+}
More information about the llvm-commits
mailing list