[llvm] [InstCombine] Match swapped form of truncating saturation clamp (PR #226614)

Craig Topper via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 25 17:48:38 PDT 2026


https://github.com/topperc created https://github.com/llvm/llvm-project/pull/226614

Extend the fold added in #189703 to also handle the inverted select:
trunc (select (icmp ugt A, DestTy_umax), sext(icmp sgt A, 0), A) -->
trunc (smin (smax (0, A), DestTy_umax))

InstCombine canonicalizes (A & NegPow2) != 0 into the ult form with
swapped select operands, but if SCCP first rewrites the compare as
icmp uge A, C, InstCombine only turns it into icmp ugt and never swaps
the select, so the original fold is missed.

While here, match the compare constant with m_APInt instead of
m_Constant + getUniqueInteger, and build TruncatedMax directly with
APInt::getLowBitsSet. Comparing the constant against TruncatedMax + 1
or TruncatedMax makes the separate zero check unnecessary.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply at anthropic.com>

>From a8590158fb7abfbf499e0c5821eca6504d9ffde6 Mon Sep 17 00:00:00 2001
From: Craig Topper <craig.topper at sifive.com>
Date: Fri, 25 Sep 2026 17:35:44 -0700
Subject: [PATCH 1/2] [InstCombine] Add tests for swapped truncating saturation
 clamp. NFC

Add tests for the inverted form of the pattern matched in visitTrunc:
trunc (select (icmp ugt A, DestTy_umax), sext(icmp sgt A, 0), A)

This form can appear when SCCP rewrites (A & NegPow2) != 0 as
icmp uge A, C before InstCombine runs.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply at anthropic.com>
---
 .../InstCombine/truncating-saturate.ll        | 99 +++++++++++++++++++
 1 file changed, 99 insertions(+)

diff --git a/llvm/test/Transforms/InstCombine/truncating-saturate.ll b/llvm/test/Transforms/InstCombine/truncating-saturate.ll
index 2eed99137a83d1..b7f7a00f12c157 100644
--- a/llvm/test/Transforms/InstCombine/truncating-saturate.ll
+++ b/llvm/test/Transforms/InstCombine/truncating-saturate.ll
@@ -781,3 +781,102 @@ entry:
   %trunc = trunc <4 x i32> %cond to <4 x i8>
   ret <4 x i8> %trunc
 }
+
+define i8 @clamp_i32_to_i8_swapped(i32 %x) {
+; CHECK-LABEL: @clamp_i32_to_i8_swapped(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ugt i32 [[X:%.*]], 255
+; CHECK-NEXT:    [[TMP0:%.*]] = icmp sgt i32 [[X]], 0
+; CHECK-NEXT:    [[SHR:%.*]] = sext i1 [[TMP0]] to i32
+; CHECK-NEXT:    [[COND:%.*]] = select i1 [[CMP]], i32 [[SHR]], i32 [[X]]
+; CHECK-NEXT:    [[TRUNC:%.*]] = trunc i32 [[COND]] to i8
+; CHECK-NEXT:    ret i8 [[TRUNC]]
+;
+entry:
+  %cmp = icmp ugt i32 %x, 255
+  %0 = icmp sgt i32 %x, 0
+  %shr = sext i1 %0 to i32
+  %cond = select i1 %cmp, i32 %shr, i32 %x
+  %trunc = trunc i32 %cond to i8
+  ret i8 %trunc
+}
+
+define i16 @clamp_i64_to_i16_swapped(i64 %x) {
+; CHECK-LABEL: @clamp_i64_to_i16_swapped(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ugt i64 [[X:%.*]], 65535
+; CHECK-NEXT:    [[TMP0:%.*]] = icmp sgt i64 [[X]], 0
+; CHECK-NEXT:    [[SHR:%.*]] = sext i1 [[TMP0]] to i64
+; CHECK-NEXT:    [[COND:%.*]] = select i1 [[CMP]], i64 [[SHR]], i64 [[X]]
+; CHECK-NEXT:    [[TRUNC:%.*]] = trunc i64 [[COND]] to i16
+; CHECK-NEXT:    ret i16 [[TRUNC]]
+;
+entry:
+  %cmp = icmp ugt i64 %x, 65535
+  %0 = icmp sgt i64 %x, 0
+  %shr = sext i1 %0 to i64
+  %cond = select i1 %cmp, i64 %shr, i64 %x
+  %trunc = trunc i64 %cond to i16
+  ret i16 %trunc
+}
+
+define i8 @no_clamp_i32_to_i8_swapped_wrong_const(i32 %x) {
+; CHECK-LABEL: @no_clamp_i32_to_i8_swapped_wrong_const(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ugt i32 [[X:%.*]], 256
+; CHECK-NEXT:    [[TMP0:%.*]] = icmp sgt i32 [[X]], 0
+; CHECK-NEXT:    [[SHR:%.*]] = sext i1 [[TMP0]] to i32
+; CHECK-NEXT:    [[COND:%.*]] = select i1 [[CMP]], i32 [[SHR]], i32 [[X]]
+; CHECK-NEXT:    [[TRUNC:%.*]] = trunc i32 [[COND]] to i8
+; CHECK-NEXT:    ret i8 [[TRUNC]]
+;
+entry:
+  %cmp = icmp ugt i32 %x, 256
+  %0 = icmp sgt i32 %x, 0
+  %shr = sext i1 %0 to i32
+  %cond = select i1 %cmp, i32 %shr, i32 %x
+  %trunc = trunc i32 %cond to i8
+  ret i8 %trunc
+}
+
+define i8 @no_clamp_i32_to_i8_swapped_wrong_arms(i32 %x) {
+; CHECK-LABEL: @no_clamp_i32_to_i8_swapped_wrong_arms(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ugt i32 [[X:%.*]], 255
+; CHECK-NEXT:    [[TMP0:%.*]] = icmp sgt i32 [[X]], 0
+; CHECK-NEXT:    [[SHR:%.*]] = sext i1 [[TMP0]] to i32
+; CHECK-NEXT:    [[COND:%.*]] = select i1 [[CMP]], i32 [[X]], i32 [[SHR]]
+; CHECK-NEXT:    [[TRUNC:%.*]] = trunc i32 [[COND]] to i8
+; CHECK-NEXT:    ret i8 [[TRUNC]]
+;
+entry:
+  %cmp = icmp ugt i32 %x, 255
+  %0 = icmp sgt i32 %x, 0
+  %shr = sext i1 %0 to i32
+  %cond = select i1 %cmp, i32 %x, i32 %shr
+  %trunc = trunc i32 %cond to i8
+  ret i8 %trunc
+}
+
+define i8 @no_clamp_i32_to_i8_swapped_multiple_use_icmp(i32 %x, ptr %p) {
+; CHECK-LABEL: @no_clamp_i32_to_i8_swapped_multiple_use_icmp(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ugt i32 [[X:%.*]], 255
+; CHECK-NEXT:    [[TMP0:%.*]] = icmp sgt i32 [[X]], 0
+; CHECK-NEXT:    [[SHR:%.*]] = sext i1 [[TMP0]] to i32
+; CHECK-NEXT:    [[COND:%.*]] = select i1 [[CMP]], i32 [[SHR]], i32 [[X]]
+; CHECK-NEXT:    [[TRUNC:%.*]] = trunc i32 [[COND]] to i8
+; CHECK-NEXT:    [[EXTRA_USE:%.*]] = sext i1 [[CMP]] to i64
+; CHECK-NEXT:    store i64 [[EXTRA_USE]], ptr [[P:%.*]], align 8
+; CHECK-NEXT:    ret i8 [[TRUNC]]
+;
+entry:
+  %cmp = icmp ugt i32 %x, 255
+  %0 = icmp sgt i32 %x, 0
+  %shr = sext i1 %0 to i32
+  %cond = select i1 %cmp, i32 %shr, i32 %x
+  %trunc = trunc i32 %cond to i8
+  %extra_use = sext i1 %cmp to i64
+  store i64 %extra_use, ptr %p, align 8
+  ret i8 %trunc
+}

>From 66949d5e10fa9152796c674f218dce4518d0a007 Mon Sep 17 00:00:00 2001
From: Craig Topper <craig.topper at sifive.com>
Date: Fri, 25 Sep 2026 17:36:35 -0700
Subject: [PATCH 2/2] [InstCombine] Match swapped form of truncating saturation
 clamp

Extend the fold added in #189703 to also handle the inverted select:
trunc (select (icmp ugt A, DestTy_umax), sext(icmp sgt A, 0), A) -->
trunc (smin (smax (0, A), DestTy_umax))

InstCombine canonicalizes (A & NegPow2) != 0 into the ult form with
swapped select operands, but if SCCP first rewrites the compare as
icmp uge A, C, InstCombine only turns it into icmp ugt and never swaps
the select, so the original fold is missed.

While here, match the compare constant with m_APInt instead of
m_Constant + getUniqueInteger, and build TruncatedMax directly with
APInt::getLowBitsSet. Comparing the constant against TruncatedMax + 1
or TruncatedMax makes the separate zero check unnecessary.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply at anthropic.com>
---
 .../InstCombine/InstCombineCasts.cpp          | 31 ++++++++++++-------
 .../InstCombine/truncating-saturate.ll        | 16 ++++------
 2 files changed, 25 insertions(+), 22 deletions(-)

diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
index 9269e681944e86..4e3061e72cfaff 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
@@ -1253,18 +1253,25 @@ Instruction *InstCombinerImpl::visitTrunc(TruncInst &Trunc) {
 
   // trunc (select(icmp_ult(A, DestTy_umax+1), A, sext(icmp_sgt(A, 0)))) -->
   // trunc (smin(smax(0, A), DestTy_umax))
-  if (SrcTy->isIntegerTy() && isPowerOf2_64(SrcTy->getPrimitiveSizeInBits()) &&
-      isPowerOf2_64(DestTy->getPrimitiveSizeInBits()) &&
-      match(Src, m_OneUse(m_Select(
-                     m_OneUse(m_SpecificICmp(ICmpInst::ICMP_ULT, m_Value(A),
-                                             m_Constant(C))),
-                     m_Deferred(A),
-                     m_OneUse(m_SExt(m_OneUse(m_SpecificICmp(
-                         ICmpInst::ICMP_SGT, m_Deferred(A), m_Zero())))))))) {
-    APInt UpperBound = C->getUniqueInteger();
-    APInt TruncatedMax = APInt::getAllOnes(DestTy->getIntegerBitWidth());
-    TruncatedMax = TruncatedMax.zext(UpperBound.getBitWidth());
-    if (!UpperBound.isZero() && UpperBound - 1 == TruncatedMax) {
+  // Also handle the inverted form:
+  // trunc (select(icmp_ugt(A, DestTy_umax), sext(icmp_sgt(A, 0)), A))
+  CmpPredicate Pred;
+  const APInt *CmpC;
+  Value *TVal, *FVal;
+  if (SrcTy->isIntegerTy() && isPowerOf2_64(SrcWidth) &&
+      isPowerOf2_64(DestWidth) &&
+      match(Src,
+            m_OneUse(m_Select(m_OneUse(m_ICmp(Pred, m_Value(A), m_APInt(CmpC))),
+                              m_Value(TVal), m_Value(FVal))))) {
+    APInt TruncatedMax = APInt::getLowBitsSet(SrcWidth, DestWidth);
+    Value *SExtVal = nullptr;
+    if (Pred == ICmpInst::ICMP_ULT && *CmpC == TruncatedMax + 1 && TVal == A)
+      SExtVal = FVal;
+    else if (Pred == ICmpInst::ICMP_UGT && *CmpC == TruncatedMax && FVal == A)
+      SExtVal = TVal;
+    if (SExtVal &&
+        match(SExtVal, m_OneUse(m_SExt(m_OneUse(m_SpecificICmp(
+                           ICmpInst::ICMP_SGT, m_Specific(A), m_Zero())))))) {
       Value *SMax = Builder.CreateIntrinsic(Intrinsic::smax, {SrcTy},
                                             {ConstantInt::get(SrcTy, 0), A});
       Value *SMin = Builder.CreateIntrinsic(
diff --git a/llvm/test/Transforms/InstCombine/truncating-saturate.ll b/llvm/test/Transforms/InstCombine/truncating-saturate.ll
index b7f7a00f12c157..454c0e9e2764fb 100644
--- a/llvm/test/Transforms/InstCombine/truncating-saturate.ll
+++ b/llvm/test/Transforms/InstCombine/truncating-saturate.ll
@@ -785,11 +785,9 @@ entry:
 define i8 @clamp_i32_to_i8_swapped(i32 %x) {
 ; CHECK-LABEL: @clamp_i32_to_i8_swapped(
 ; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[CMP:%.*]] = icmp ugt i32 [[X:%.*]], 255
-; CHECK-NEXT:    [[TMP0:%.*]] = icmp sgt i32 [[X]], 0
-; CHECK-NEXT:    [[SHR:%.*]] = sext i1 [[TMP0]] to i32
-; CHECK-NEXT:    [[COND:%.*]] = select i1 [[CMP]], i32 [[SHR]], i32 [[X]]
-; CHECK-NEXT:    [[TRUNC:%.*]] = trunc i32 [[COND]] to i8
+; CHECK-NEXT:    [[TMP0:%.*]] = call i32 @llvm.smax.i32(i32 [[X:%.*]], i32 0)
+; CHECK-NEXT:    [[TMP1:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP0]], i32 255)
+; CHECK-NEXT:    [[TRUNC:%.*]] = trunc nuw i32 [[TMP1]] to i8
 ; CHECK-NEXT:    ret i8 [[TRUNC]]
 ;
 entry:
@@ -804,11 +802,9 @@ entry:
 define i16 @clamp_i64_to_i16_swapped(i64 %x) {
 ; CHECK-LABEL: @clamp_i64_to_i16_swapped(
 ; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[CMP:%.*]] = icmp ugt i64 [[X:%.*]], 65535
-; CHECK-NEXT:    [[TMP0:%.*]] = icmp sgt i64 [[X]], 0
-; CHECK-NEXT:    [[SHR:%.*]] = sext i1 [[TMP0]] to i64
-; CHECK-NEXT:    [[COND:%.*]] = select i1 [[CMP]], i64 [[SHR]], i64 [[X]]
-; CHECK-NEXT:    [[TRUNC:%.*]] = trunc i64 [[COND]] to i16
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[X:%.*]], i64 0)
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 65535)
+; CHECK-NEXT:    [[TRUNC:%.*]] = trunc nuw i64 [[TMP1]] to i16
 ; CHECK-NEXT:    ret i16 [[TRUNC]]
 ;
 entry:



More information about the llvm-commits mailing list