[llvm] [InstCombine] Fold uitofp of masked truncations (PR #222592)
Harrison Hao via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 10 03:33:18 PDT 2026
https://github.com/harrisonGPU created https://github.com/llvm/llvm-project/pull/222592
Fold an unsigned integer-to-floating-point conversion of a masked
truncation by widening the mask and eliminating the truncation:
`uitofp((trunc X) & C)` → `uitofp(X & zext(C))`
Apply the fold when the intermediate instructions have one use and
widening the integer operation is desirable.
This reduces the instruction count on both AMDGPU and NVPTX.
For AMDGPU example: https://godbolt.org/z/rTYhM61ce
For NVPTX example: https://godbolt.org/z/fWhsrne19
>From e5673d174e53b750efba8fa7f7918298f77f3aae Mon Sep 17 00:00:00 2001
From: Harrison Hao <tsworld1314 at gmail.com>
Date: Thu, 10 Sep 2026 18:24:27 +0800
Subject: [PATCH] [InstCombine] Fold uitofp of masked truncations
---
.../InstCombine/InstCombineCasts.cpp | 17 ++++
llvm/test/Transforms/InstCombine/uitofp.ll | 77 +++++++++++++++++++
2 files changed, 94 insertions(+)
create mode 100644 llvm/test/Transforms/InstCombine/uitofp.ll
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
index 54d811fa3e923..7fa0716bdd9e9 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
@@ -2630,6 +2630,23 @@ Instruction *InstCombinerImpl::visitUIToFP(CastInst &CI) {
CI.setNonNeg();
return &CI;
}
+
+ // uitofp (and (trunc X), Mask) --> uitofp (and X, zext(Mask))
+ Value *Src = CI.getOperand(0);
+ Value *X;
+ Constant *Mask;
+ if (match(Src, m_OneUse(m_And(m_OneUse(m_Trunc(m_Value(X))),
+ m_ImmConstant(Mask)))) &&
+ shouldChangeType(Src->getType()->getScalarSizeInBits(),
+ X->getType()->getScalarSizeInBits())) {
+ Value *MaskedX =
+ Builder.CreateAnd(X, Builder.CreateZExt(Mask, X->getType()));
+ auto *NewUIToFP =
+ CastInst::Create(Instruction::UIToFP, MaskedX, CI.getType());
+ NewUIToFP->setNonNeg(CI.hasNonNeg());
+ return NewUIToFP;
+ }
+
return nullptr;
}
diff --git a/llvm/test/Transforms/InstCombine/uitofp.ll b/llvm/test/Transforms/InstCombine/uitofp.ll
new file mode 100644
index 0000000000000..5f0fb1b0c32aa
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/uitofp.ll
@@ -0,0 +1,77 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt < %s -passes=instcombine -S | FileCheck %s
+
+target datalayout = "n32:64"
+
+ at external_mask = external global i8
+
+define half @uitofp_trunc_and_mask_scalar(i32 %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_scalar(
+; CHECK-NEXT: [[TMP1:%.*]] = and i32 [[X:%.*]], 255
+; CHECK-NEXT: [[RESULT:%.*]] = uitofp nneg i32 [[TMP1]] to half
+; CHECK-NEXT: ret half [[RESULT]]
+;
+ %trunc = trunc i32 %x to i16
+ %masked = and i16 %trunc, 255
+ %result = uitofp nneg i16 %masked to half
+ ret half %result
+}
+
+define <2 x half> @uitofp_trunc_and_mask_vector(<2 x i32> %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_vector(
+; CHECK-NEXT: [[TMP1:%.*]] = and <2 x i32> [[X:%.*]], splat (i32 255)
+; CHECK-NEXT: [[RESULT:%.*]] = uitofp nneg <2 x i32> [[TMP1]] to <2 x half>
+; CHECK-NEXT: ret <2 x half> [[RESULT]]
+;
+ %trunc = trunc <2 x i32> %x to <2 x i16>
+ %masked = and <2 x i16> %trunc, splat (i16 255)
+ %result = uitofp nneg <2 x i16> %masked to <2 x half>
+ ret <2 x half> %result
+}
+
+define <2 x half> @uitofp_trunc_and_constexpr_mask(<2 x i32> %x) {
+; CHECK-LABEL: @uitofp_trunc_and_constexpr_mask(
+; CHECK-NEXT: [[TRUNC:%.*]] = trunc <2 x i32> [[X:%.*]] to <2 x i16>
+; CHECK-NEXT: [[MASKED:%.*]] = and <2 x i16> [[TRUNC]], <i16 ptrtoint (ptr @external_mask to i16), i16 ptrtoint (ptr @external_mask to i16)>
+; CHECK-NEXT: [[RESULT:%.*]] = uitofp nneg <2 x i16> [[MASKED]] to <2 x half>
+; CHECK-NEXT: ret <2 x half> [[RESULT]]
+;
+ %trunc = trunc <2 x i32> %x to <2 x i16>
+ %masked = and <2 x i16> %trunc, splat (i16 ptrtoint (ptr @external_mask to i16))
+ %result = uitofp nneg <2 x i16> %masked to <2 x half>
+ ret <2 x half> %result
+}
+
+define { <2 x half>, <2 x i16> } @uitofp_trunc_and_mask_multiuse_and(<2 x i32> %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_multiuse_and(
+; CHECK-NEXT: [[TRUNC:%.*]] = trunc <2 x i32> [[X:%.*]] to <2 x i16>
+; CHECK-NEXT: [[MASKED:%.*]] = and <2 x i16> [[TRUNC]], splat (i16 255)
+; CHECK-NEXT: [[RESULT:%.*]] = uitofp nneg <2 x i16> [[MASKED]] to <2 x half>
+; CHECK-NEXT: [[RET0:%.*]] = insertvalue { <2 x half>, <2 x i16> } poison, <2 x half> [[RESULT]], 0
+; CHECK-NEXT: [[RET1:%.*]] = insertvalue { <2 x half>, <2 x i16> } [[RET0]], <2 x i16> [[MASKED]], 1
+; CHECK-NEXT: ret { <2 x half>, <2 x i16> } [[RET1]]
+;
+ %trunc = trunc <2 x i32> %x to <2 x i16>
+ %masked = and <2 x i16> %trunc, splat (i16 255)
+ %result = uitofp nneg <2 x i16> %masked to <2 x half>
+ %ret0 = insertvalue { <2 x half>, <2 x i16> } poison, <2 x half> %result, 0
+ %ret1 = insertvalue { <2 x half>, <2 x i16> } %ret0, <2 x i16> %masked, 1
+ ret { <2 x half>, <2 x i16> } %ret1
+}
+
+define { <2 x half>, <2 x i16> } @uitofp_trunc_and_mask_multiuse_trunc(<2 x i32> %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_multiuse_trunc(
+; CHECK-NEXT: [[TRUNC:%.*]] = trunc <2 x i32> [[X:%.*]] to <2 x i16>
+; CHECK-NEXT: [[MASKED:%.*]] = and <2 x i16> [[TRUNC]], splat (i16 255)
+; CHECK-NEXT: [[RESULT:%.*]] = uitofp nneg <2 x i16> [[MASKED]] to <2 x half>
+; CHECK-NEXT: [[RET0:%.*]] = insertvalue { <2 x half>, <2 x i16> } poison, <2 x half> [[RESULT]], 0
+; CHECK-NEXT: [[RET1:%.*]] = insertvalue { <2 x half>, <2 x i16> } [[RET0]], <2 x i16> [[TRUNC]], 1
+; CHECK-NEXT: ret { <2 x half>, <2 x i16> } [[RET1]]
+;
+ %trunc = trunc <2 x i32> %x to <2 x i16>
+ %masked = and <2 x i16> %trunc, splat (i16 255)
+ %result = uitofp nneg <2 x i16> %masked to <2 x half>
+ %ret0 = insertvalue { <2 x half>, <2 x i16> } poison, <2 x half> %result, 0
+ %ret1 = insertvalue { <2 x half>, <2 x i16> } %ret0, <2 x i16> %trunc, 1
+ ret { <2 x half>, <2 x i16> } %ret1
+}
More information about the llvm-commits
mailing list