[llvm] [InstCombine] Fold uitofp of masked truncations (PR #222592)

Harrison Hao via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 10 03:33:18 PDT 2026


https://github.com/harrisonGPU created https://github.com/llvm/llvm-project/pull/222592

Fold an unsigned integer-to-floating-point conversion of a masked
truncation by widening the mask and eliminating the truncation:

`uitofp((trunc X) & C)` → `uitofp(X & zext(C))`

Apply the fold when the intermediate instructions have one use and
widening the integer operation is desirable.

This reduces the instruction count on both AMDGPU and NVPTX.

For AMDGPU example: https://godbolt.org/z/rTYhM61ce
For NVPTX example: https://godbolt.org/z/fWhsrne19

>From e5673d174e53b750efba8fa7f7918298f77f3aae Mon Sep 17 00:00:00 2001
From: Harrison Hao <tsworld1314 at gmail.com>
Date: Thu, 10 Sep 2026 18:24:27 +0800
Subject: [PATCH] [InstCombine] Fold uitofp of masked truncations

---
 .../InstCombine/InstCombineCasts.cpp          | 17 ++++
 llvm/test/Transforms/InstCombine/uitofp.ll    | 77 +++++++++++++++++++
 2 files changed, 94 insertions(+)
 create mode 100644 llvm/test/Transforms/InstCombine/uitofp.ll

diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
index 54d811fa3e923..7fa0716bdd9e9 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
@@ -2630,6 +2630,23 @@ Instruction *InstCombinerImpl::visitUIToFP(CastInst &CI) {
     CI.setNonNeg();
     return &CI;
   }
+
+  // uitofp (and (trunc X), Mask) --> uitofp (and X, zext(Mask))
+  Value *Src = CI.getOperand(0);
+  Value *X;
+  Constant *Mask;
+  if (match(Src, m_OneUse(m_And(m_OneUse(m_Trunc(m_Value(X))),
+                                m_ImmConstant(Mask)))) &&
+      shouldChangeType(Src->getType()->getScalarSizeInBits(),
+                       X->getType()->getScalarSizeInBits())) {
+    Value *MaskedX =
+        Builder.CreateAnd(X, Builder.CreateZExt(Mask, X->getType()));
+    auto *NewUIToFP =
+        CastInst::Create(Instruction::UIToFP, MaskedX, CI.getType());
+    NewUIToFP->setNonNeg(CI.hasNonNeg());
+    return NewUIToFP;
+  }
+
   return nullptr;
 }
 
diff --git a/llvm/test/Transforms/InstCombine/uitofp.ll b/llvm/test/Transforms/InstCombine/uitofp.ll
new file mode 100644
index 0000000000000..5f0fb1b0c32aa
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/uitofp.ll
@@ -0,0 +1,77 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt < %s -passes=instcombine -S | FileCheck %s
+
+target datalayout = "n32:64"
+
+ at external_mask = external global i8
+
+define half @uitofp_trunc_and_mask_scalar(i32 %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_scalar(
+; CHECK-NEXT:    [[TMP1:%.*]] = and i32 [[X:%.*]], 255
+; CHECK-NEXT:    [[RESULT:%.*]] = uitofp nneg i32 [[TMP1]] to half
+; CHECK-NEXT:    ret half [[RESULT]]
+;
+  %trunc = trunc i32 %x to i16
+  %masked = and i16 %trunc, 255
+  %result = uitofp nneg i16 %masked to half
+  ret half %result
+}
+
+define <2 x half> @uitofp_trunc_and_mask_vector(<2 x i32> %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_vector(
+; CHECK-NEXT:    [[TMP1:%.*]] = and <2 x i32> [[X:%.*]], splat (i32 255)
+; CHECK-NEXT:    [[RESULT:%.*]] = uitofp nneg <2 x i32> [[TMP1]] to <2 x half>
+; CHECK-NEXT:    ret <2 x half> [[RESULT]]
+;
+  %trunc = trunc <2 x i32> %x to <2 x i16>
+  %masked = and <2 x i16> %trunc, splat (i16 255)
+  %result = uitofp nneg <2 x i16> %masked to <2 x half>
+  ret <2 x half> %result
+}
+
+define <2 x half> @uitofp_trunc_and_constexpr_mask(<2 x i32> %x) {
+; CHECK-LABEL: @uitofp_trunc_and_constexpr_mask(
+; CHECK-NEXT:    [[TRUNC:%.*]] = trunc <2 x i32> [[X:%.*]] to <2 x i16>
+; CHECK-NEXT:    [[MASKED:%.*]] = and <2 x i16> [[TRUNC]], <i16 ptrtoint (ptr @external_mask to i16), i16 ptrtoint (ptr @external_mask to i16)>
+; CHECK-NEXT:    [[RESULT:%.*]] = uitofp nneg <2 x i16> [[MASKED]] to <2 x half>
+; CHECK-NEXT:    ret <2 x half> [[RESULT]]
+;
+  %trunc = trunc <2 x i32> %x to <2 x i16>
+  %masked = and <2 x i16> %trunc, splat (i16 ptrtoint (ptr @external_mask to i16))
+  %result = uitofp nneg <2 x i16> %masked to <2 x half>
+  ret <2 x half> %result
+}
+
+define { <2 x half>, <2 x i16> } @uitofp_trunc_and_mask_multiuse_and(<2 x i32> %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_multiuse_and(
+; CHECK-NEXT:    [[TRUNC:%.*]] = trunc <2 x i32> [[X:%.*]] to <2 x i16>
+; CHECK-NEXT:    [[MASKED:%.*]] = and <2 x i16> [[TRUNC]], splat (i16 255)
+; CHECK-NEXT:    [[RESULT:%.*]] = uitofp nneg <2 x i16> [[MASKED]] to <2 x half>
+; CHECK-NEXT:    [[RET0:%.*]] = insertvalue { <2 x half>, <2 x i16> } poison, <2 x half> [[RESULT]], 0
+; CHECK-NEXT:    [[RET1:%.*]] = insertvalue { <2 x half>, <2 x i16> } [[RET0]], <2 x i16> [[MASKED]], 1
+; CHECK-NEXT:    ret { <2 x half>, <2 x i16> } [[RET1]]
+;
+  %trunc = trunc <2 x i32> %x to <2 x i16>
+  %masked = and <2 x i16> %trunc, splat (i16 255)
+  %result = uitofp nneg <2 x i16> %masked to <2 x half>
+  %ret0 = insertvalue { <2 x half>, <2 x i16> } poison, <2 x half> %result, 0
+  %ret1 = insertvalue { <2 x half>, <2 x i16> } %ret0, <2 x i16> %masked, 1
+  ret { <2 x half>, <2 x i16> } %ret1
+}
+
+define { <2 x half>, <2 x i16> } @uitofp_trunc_and_mask_multiuse_trunc(<2 x i32> %x) {
+; CHECK-LABEL: @uitofp_trunc_and_mask_multiuse_trunc(
+; CHECK-NEXT:    [[TRUNC:%.*]] = trunc <2 x i32> [[X:%.*]] to <2 x i16>
+; CHECK-NEXT:    [[MASKED:%.*]] = and <2 x i16> [[TRUNC]], splat (i16 255)
+; CHECK-NEXT:    [[RESULT:%.*]] = uitofp nneg <2 x i16> [[MASKED]] to <2 x half>
+; CHECK-NEXT:    [[RET0:%.*]] = insertvalue { <2 x half>, <2 x i16> } poison, <2 x half> [[RESULT]], 0
+; CHECK-NEXT:    [[RET1:%.*]] = insertvalue { <2 x half>, <2 x i16> } [[RET0]], <2 x i16> [[TRUNC]], 1
+; CHECK-NEXT:    ret { <2 x half>, <2 x i16> } [[RET1]]
+;
+  %trunc = trunc <2 x i32> %x to <2 x i16>
+  %masked = and <2 x i16> %trunc, splat (i16 255)
+  %result = uitofp nneg <2 x i16> %masked to <2 x half>
+  %ret0 = insertvalue { <2 x half>, <2 x i16> } poison, <2 x half> %result, 0
+  %ret1 = insertvalue { <2 x half>, <2 x i16> } %ret0, <2 x i16> %trunc, 1
+  ret { <2 x half>, <2 x i16> } %ret1
+}



More information about the llvm-commits mailing list