[llvm] [InstCombine] Optimize `zext(trunc nuw x)` (PR #225206)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 17:55:57 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: peterbell10

<details>
<summary>Changes</summary>

We already have the equivalent transformation for `sext(trunc nsw x)`, this factors it into `commonCastTransforms` and generalizes it to work for `zext` with `nuw`.

Alive2 proofs for `zext`: https://alive2.llvm.org/ce/z/r-2bfQ
Alive2 proofs for `sext`: https://alive2.llvm.org/ce/z/VF6AYg

Assisted-by: GPT-6 Astra

---

Patch is 53.89 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/225206.diff


5 Files Affected:

- (modified) llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp (+19-7) 
- (modified) llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll (+40) 
- (modified) llvm/test/Transforms/InstCombine/trunc.ll (+2-4) 
- (modified) llvm/test/Transforms/InstCombine/zext.ll (+104) 
- (modified) llvm/test/Transforms/PhaseOrdering/X86/avg.ll (+263-146) 


``````````diff
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
index 97defc5e3ddd3..274eed449dc59 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
@@ -239,6 +239,23 @@ Instruction *InstCombinerImpl::commonCastTransforms(CastInst &CI) {
         replaceAllDbgUsesWith(*CSrc, *Res, CI, DT);
       return Res;
     }
+
+    // A no-wrap trunc followed by the corresponding extension preserves the
+    // value, so cast directly to the final type.
+    bool IsZExt = isa<ZExtInst>(CI);
+    bool IsSExt = isa<SExtInst>(CI);
+    if (auto *Trunc = dyn_cast<TruncInst>(CSrc);
+        Trunc && ((IsZExt && Trunc->hasNoUnsignedWrap()) ||
+                  (IsSExt && Trunc->hasNoSignedWrap()))) {
+      auto *Res = CastInst::CreateIntegerCast(Trunc->getOperand(0), Ty, IsSExt);
+      if (auto *ResTrunc = dyn_cast<TruncInst>(Res)) {
+        ResTrunc->setHasNoUnsignedWrap(Trunc->hasNoUnsignedWrap());
+        ResTrunc->setHasNoSignedWrap(Trunc->hasNoSignedWrap());
+      } else if (auto *ResZExt = dyn_cast<ZExtInst>(Res)) {
+        ResZExt->setNonNeg(true);
+      }
+      return Res;
+    }
   }
 
   if (auto *Sel = dyn_cast<SelectInst>(Src)) {
@@ -1967,13 +1984,8 @@ Instruction *InstCombinerImpl::visitSExt(SExtInst &Sext) {
     // If the input has more sign bits than bits truncated, then convert
     // directly to final type.
     unsigned XBitSize = X->getType()->getScalarSizeInBits();
-    bool HasNSW = cast<TruncInst>(Src)->hasNoSignedWrap();
-    if (HasNSW || (ComputeNumSignBits(X, &Sext) > XBitSize - SrcBitSize)) {
-      auto *Res = CastInst::CreateIntegerCast(X, DestTy, /* isSigned */ true);
-      if (auto *ResTrunc = dyn_cast<TruncInst>(Res); ResTrunc && HasNSW)
-        ResTrunc->setHasNoSignedWrap(true);
-      return Res;
-    }
+    if (ComputeNumSignBits(X, &Sext) > XBitSize - SrcBitSize)
+      return CastInst::CreateIntegerCast(X, DestTy, /* isSigned */ true);
 
     // If input is a trunc from the destination type, then convert into shifts.
     if (Src->hasOneUse() && X->getType() == DestTy) {
diff --git a/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll b/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll
index 4128a15d8d7ce..d6fb3a10fccd6 100644
--- a/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll
+++ b/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll
@@ -233,3 +233,43 @@ define i32 @same_source_not_matching_signbits_extra_use(i32 %x) {
   %c = sext i8 %b to i32
   ret i32 %c
 }
+
+define i32 @sext_trunc_widen(i8 %x) {
+; CHECK-LABEL: @sext_trunc_widen(
+; CHECK-NEXT:    [[RESULT:%.*]] = sext i8 [[X:%.*]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc nsw i8 %x to i1
+  %result = sext i1 %narrow to i32
+  ret i32 %result
+}
+
+define i8 @sext_trunc_same(i8 %x) {
+; CHECK-LABEL: @sext_trunc_same(
+; CHECK-NEXT:    ret i8 [[X:%.*]]
+;
+  %narrow = trunc nsw i8 %x to i1
+  %result = sext i1 %narrow to i8
+  ret i8 %result
+}
+
+define i32 @sext_trunc_narrow(i64 %x) {
+; CHECK-LABEL: @sext_trunc_narrow(
+; CHECK-NEXT:    [[RESULT:%.*]] = trunc nsw i64 [[X:%.*]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc nsw i64 %x to i1
+  %result = sext i1 %narrow to i32
+  ret i32 %result
+}
+
+define i32 @sext_trunc_nuw(i8 %x) {
+; CHECK-LABEL: @sext_trunc_nuw(
+; CHECK-NEXT:    [[NARROW:%.*]] = trunc nuw i8 [[X:%.*]] to i1
+; CHECK-NEXT:    [[RESULT:%.*]] = sext i1 [[NARROW]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc nuw i8 %x to i1
+  %result = sext i1 %narrow to i32
+  ret i32 %result
+}
diff --git a/llvm/test/Transforms/InstCombine/trunc.ll b/llvm/test/Transforms/InstCombine/trunc.ll
index 91f637e89069f..6f99bf444ccb4 100644
--- a/llvm/test/Transforms/InstCombine/trunc.ll
+++ b/llvm/test/Transforms/InstCombine/trunc.ll
@@ -1406,8 +1406,7 @@ define i32 @neg_zext_i32_trunc_nsw_i8(i16 %x, i32 %y) {
 
 define i16 @zext_i16_trunc_nuw_nsw_i8(i32 %x) {
 ; CHECK-LABEL: @zext_i16_trunc_nuw_nsw_i8(
-; CHECK-NEXT:    [[C:%.*]] = trunc nuw nsw i32 [[X:%.*]] to i16
-; CHECK-NEXT:    [[E:%.*]] = and i16 [[C]], 255
+; CHECK-NEXT:    [[E:%.*]] = trunc nuw nsw i32 [[X:%.*]] to i16
 ; CHECK-NEXT:    ret i16 [[E]]
 ;
   %c = trunc nuw nsw i32 %x to i8
@@ -1428,8 +1427,7 @@ define i16 @zext_i16_trunc_nsw_i8(i32 %x) {
 
 define i16 @zext_i16_trunc_nuw_i8(i32 %x) {
 ; CHECK-LABEL: @zext_i16_trunc_nuw_i8(
-; CHECK-NEXT:    [[C:%.*]] = trunc nuw i32 [[X:%.*]] to i16
-; CHECK-NEXT:    [[E:%.*]] = and i16 [[C]], 255
+; CHECK-NEXT:    [[E:%.*]] = trunc nuw i32 [[X:%.*]] to i16
 ; CHECK-NEXT:    ret i16 [[E]]
 ;
   %c = trunc nuw i32 %x to i8
diff --git a/llvm/test/Transforms/InstCombine/zext.ll b/llvm/test/Transforms/InstCombine/zext.ll
index 3fb1e77ae2335..e98710d6b43d1 100644
--- a/llvm/test/Transforms/InstCombine/zext.ll
+++ b/llvm/test/Transforms/InstCombine/zext.ll
@@ -1080,3 +1080,107 @@ define <2 x i8> @zext_or_trunc_nuw_vec(<2 x i8> %x, <2 x i4> %y) {
   %zext = zext <2 x i4> %or to <2 x i8>
   ret <2 x i8> %zext
 }
+
+define i32 @zext_trunc_widen(i8 %x) {
+; CHECK-LABEL: @zext_trunc_widen(
+; CHECK-NEXT:    [[RESULT:%.*]] = zext nneg i8 [[X:%.*]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc nuw i8 %x to i1
+  %result = zext i1 %narrow to i32
+  ret i32 %result
+}
+
+define i64 @zext_trunc_native(i8 %x) {
+; CHECK-LABEL: @zext_trunc_native(
+; CHECK-NEXT:    [[RESULT:%.*]] = zext nneg i8 [[X:%.*]] to i64
+; CHECK-NEXT:    ret i64 [[RESULT]]
+;
+  %narrow = trunc nuw i8 %x to i1
+  %result = zext i1 %narrow to i64
+  ret i64 %result
+}
+
+define i8 @zext_trunc_same(i8 %x) {
+; CHECK-LABEL: @zext_trunc_same(
+; CHECK-NEXT:    ret i8 [[X:%.*]]
+;
+  %narrow = trunc nuw i8 %x to i1
+  %result = zext i1 %narrow to i8
+  ret i8 %result
+}
+
+define i32 @zext_trunc_narrow(i64 %x) {
+; CHECK-LABEL: @zext_trunc_narrow(
+; CHECK-NEXT:    [[RESULT:%.*]] = trunc nuw i64 [[X:%.*]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc nuw i64 %x to i1
+  %result = zext i1 %narrow to i32
+  ret i32 %result
+}
+
+define i32 @zext_trunc_no_flag(i8 %x) {
+; CHECK-LABEL: @zext_trunc_no_flag(
+; CHECK-NEXT:    [[NARROW_MASK:%.*]] = and i8 [[X:%.*]], 1
+; CHECK-NEXT:    [[RESULT:%.*]] = zext nneg i8 [[NARROW_MASK]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc i8 %x to i1
+  %result = zext i1 %narrow to i32
+  ret i32 %result
+}
+
+define i32 @zext_trunc_nsw(i8 %x) {
+; CHECK-LABEL: @zext_trunc_nsw(
+; CHECK-NEXT:    [[NARROW_MASK:%.*]] = and i8 [[X:%.*]], 1
+; CHECK-NEXT:    [[RESULT:%.*]] = zext nneg i8 [[NARROW_MASK]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc nsw i8 %x to i1
+  %result = zext i1 %narrow to i32
+  ret i32 %result
+}
+
+define <2 x i32> @zext_trunc_nuw_vec(<2 x i8> %x) {
+; CHECK-LABEL: @zext_trunc_nuw_vec(
+; CHECK-NEXT:    [[RESULT:%.*]] = zext nneg <2 x i8> [[X:%.*]] to <2 x i32>
+; CHECK-NEXT:    ret <2 x i32> [[RESULT]]
+;
+  %narrow = trunc nuw <2 x i8> %x to <2 x i1>
+  %result = zext <2 x i1> %narrow to <2 x i32>
+  ret <2 x i32> %result
+}
+
+define <vscale x 2 x i32> @zext_trunc_nuw_scalable(<vscale x 2 x i8> %x) {
+; CHECK-LABEL: @zext_trunc_nuw_scalable(
+; CHECK-NEXT:    [[RESULT:%.*]] = zext nneg <vscale x 2 x i8> [[X:%.*]] to <vscale x 2 x i32>
+; CHECK-NEXT:    ret <vscale x 2 x i32> [[RESULT]]
+;
+  %narrow = trunc nuw <vscale x 2 x i8> %x to <vscale x 2 x i1>
+  %result = zext <vscale x 2 x i1> %narrow to <vscale x 2 x i32>
+  ret <vscale x 2 x i32> %result
+}
+
+define <2 x i32> @zext_trunc_nuw_vec_narrow(<2 x i64> %x) {
+; CHECK-LABEL: @zext_trunc_nuw_vec_narrow(
+; CHECK-NEXT:    [[RESULT:%.*]] = trunc nuw <2 x i64> [[X:%.*]] to <2 x i32>
+; CHECK-NEXT:    ret <2 x i32> [[RESULT]]
+;
+  %narrow = trunc nuw <2 x i64> %x to <2 x i1>
+  %result = zext <2 x i1> %narrow to <2 x i32>
+  ret <2 x i32> %result
+}
+
+define i32 @zext_trunc_nuw_multi_use(i8 %x) {
+; CHECK-LABEL: @zext_trunc_nuw_multi_use(
+; CHECK-NEXT:    [[NARROW:%.*]] = trunc nuw i8 [[X:%.*]] to i1
+; CHECK-NEXT:    call void @use1(i1 [[NARROW]])
+; CHECK-NEXT:    [[RESULT:%.*]] = zext nneg i8 [[X]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc nuw i8 %x to i1
+  call void @use1(i1 %narrow)
+  %result = zext i1 %narrow to i32
+  ret i32 %result
+}
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
index 8f8899d285ad3..7abb52200084a 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
@@ -44,6 +44,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE2-NEXT:    [[TMP26:%.*]] = lshr <2 x i64> [[TMP25]], splat (i64 1)
 ; SSE2-NEXT:    [[TMP27:%.*]] = add nuw nsw <2 x i16> [[TMP20]], splat (i16 1)
 ; SSE2-NEXT:    [[TMP28:%.*]] = add nuw nsw <2 x i16> [[TMP27]], [[TMP23]]
+; SSE2-NEXT:    [[TMP50:%.*]] = lshr <2 x i16> [[TMP28]], splat (i16 1)
 ; SSE2-NEXT:    [[TMP29:%.*]] = and <2 x i64> [[TMP3]], splat (i64 255)
 ; SSE2-NEXT:    [[TMP30:%.*]] = and <2 x i64> [[TMP11]], splat (i64 255)
 ; SSE2-NEXT:    [[TMP31:%.*]] = add nuw nsw <2 x i64> [[TMP29]], splat (i64 1)
@@ -68,25 +69,24 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE2-NEXT:    [[TMP75:%.*]] = shl nuw <2 x i64> [[TMP74]], splat (i64 55)
 ; SSE2-NEXT:    [[TMP76:%.*]] = and <2 x i64> [[TMP75]], splat (i64 -72057594037927936)
 ; SSE2-NEXT:    [[TMP77:%.*]] = shl nuw nsw <2 x i64> [[TMP72]], splat (i64 47)
-; SSE2-NEXT:    [[TMP78:%.*]] = and <2 x i64> [[TMP77]], splat (i64 71776119061217280)
-; SSE2-NEXT:    [[TMP52:%.*]] = or disjoint <2 x i64> [[TMP76]], [[TMP78]]
+; SSE2-NEXT:    [[TMP53:%.*]] = and <2 x i64> [[TMP77]], splat (i64 143833713099145216)
+; SSE2-NEXT:    [[TMP55:%.*]] = or <2 x i64> [[TMP76]], [[TMP53]]
 ; SSE2-NEXT:    [[TMP49:%.*]] = shl nuw nsw <2 x i64> [[TMP44]], splat (i64 39)
-; SSE2-NEXT:    [[TMP50:%.*]] = and <2 x i64> [[TMP49]], splat (i64 280375465082880)
-; SSE2-NEXT:    [[TMP53:%.*]] = or disjoint <2 x i64> [[TMP52]], [[TMP50]]
+; SSE2-NEXT:    [[TMP56:%.*]] = and <2 x i64> [[TMP49]], splat (i64 561850441793536)
+; SSE2-NEXT:    [[TMP58:%.*]] = or <2 x i64> [[TMP55]], [[TMP56]]
 ; SSE2-NEXT:    [[TMP54:%.*]] = shl nuw nsw <2 x i64> [[TMP40]], splat (i64 31)
-; SSE2-NEXT:    [[TMP55:%.*]] = and <2 x i64> [[TMP54]], splat (i64 1095216660480)
-; SSE2-NEXT:    [[TMP56:%.*]] = or disjoint <2 x i64> [[TMP53]], [[TMP55]]
+; SSE2-NEXT:    [[TMP59:%.*]] = and <2 x i64> [[TMP54]], splat (i64 2194728288256)
+; SSE2-NEXT:    [[TMP61:%.*]] = or <2 x i64> [[TMP58]], [[TMP59]]
 ; SSE2-NEXT:    [[TMP57:%.*]] = shl nuw nsw <2 x i64> [[TMP36]], splat (i64 23)
-; SSE2-NEXT:    [[TMP58:%.*]] = and <2 x i64> [[TMP57]], splat (i64 4278190080)
-; SSE2-NEXT:    [[TMP59:%.*]] = or disjoint <2 x i64> [[TMP56]], [[TMP58]]
+; SSE2-NEXT:    [[TMP62:%.*]] = and <2 x i64> [[TMP57]], splat (i64 8573157376)
+; SSE2-NEXT:    [[TMP63:%.*]] = or <2 x i64> [[TMP61]], [[TMP62]]
 ; SSE2-NEXT:    [[TMP60:%.*]] = shl nuw nsw <2 x i64> [[TMP32]], splat (i64 15)
-; SSE2-NEXT:    [[TMP61:%.*]] = and <2 x i64> [[TMP60]], splat (i64 16711680)
-; SSE2-NEXT:    [[TMP62:%.*]] = shl nuw <2 x i16> [[TMP28]], splat (i16 7)
-; SSE2-NEXT:    [[TMP63:%.*]] = or disjoint <2 x i64> [[TMP59]], [[TMP61]]
-; SSE2-NEXT:    [[TMP64:%.*]] = and <2 x i16> [[TMP62]], splat (i16 -256)
-; SSE2-NEXT:    [[TMP65:%.*]] = zext <2 x i16> [[TMP64]] to <2 x i64>
+; SSE2-NEXT:    [[TMP65:%.*]] = and <2 x i64> [[TMP60]], splat (i64 33488896)
+; SSE2-NEXT:    [[TMP78:%.*]] = zext nneg <2 x i16> [[TMP50]] to <2 x i64>
+; SSE2-NEXT:    [[TMP79:%.*]] = shl nuw nsw <2 x i64> [[TMP78]], splat (i64 8)
 ; SSE2-NEXT:    [[TMP66:%.*]] = or <2 x i64> [[TMP63]], [[TMP65]]
-; SSE2-NEXT:    [[TMP67:%.*]] = or <2 x i64> [[TMP66]], [[TMP26]]
+; SSE2-NEXT:    [[TMP80:%.*]] = or <2 x i64> [[TMP66]], [[TMP79]]
+; SSE2-NEXT:    [[TMP67:%.*]] = or <2 x i64> [[TMP80]], [[TMP26]]
 ; SSE2-NEXT:    [[TMP68:%.*]] = extractelement <2 x i64> [[TMP67]], i64 0
 ; SSE2-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP68]], 0
 ; SSE2-NEXT:    [[TMP69:%.*]] = extractelement <2 x i64> [[TMP67]], i64 1
@@ -134,6 +134,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE4-NEXT:    [[SHR:%.*]] = lshr i64 [[ADD5]], 1
 ; SSE4-NEXT:    [[ADD_1:%.*]] = add nuw nsw i16 [[TMP1]], 1
 ; SSE4-NEXT:    [[ADD5_1:%.*]] = add nuw nsw i16 [[ADD_1]], [[TMP5]]
+; SSE4-NEXT:    [[SHR_1:%.*]] = lshr i16 [[ADD5_1]], 1
 ; SSE4-NEXT:    [[CONV1_6:%.*]] = and i64 [[A_SROA_7_0_EXTRACT_SHIFT]], 255
 ; SSE4-NEXT:    [[CONV4_6:%.*]] = and i64 [[B_SROA_7_0_EXTRACT_SHIFT]], 255
 ; SSE4-NEXT:    [[ADD_6:%.*]] = add nuw nsw i64 [[CONV1_6]], 1
@@ -147,6 +148,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE4-NEXT:    [[SHR_8:%.*]] = lshr i64 [[ADD5_8]], 1
 ; SSE4-NEXT:    [[ADD_9:%.*]] = add nuw nsw i16 [[TMP3]], 1
 ; SSE4-NEXT:    [[ADD5_9:%.*]] = add nuw nsw i16 [[ADD_9]], [[TMP7]]
+; SSE4-NEXT:    [[SHR_9:%.*]] = lshr i16 [[ADD5_9]], 1
 ; SSE4-NEXT:    [[CONV1_14:%.*]] = and i64 [[A_SROA_16_8_EXTRACT_SHIFT]], 255
 ; SSE4-NEXT:    [[CONV4_14:%.*]] = and i64 [[B_SROA_16_8_EXTRACT_SHIFT]], 255
 ; SSE4-NEXT:    [[ADD_14:%.*]] = add nuw nsw i64 [[CONV1_14]], 1
@@ -156,7 +158,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE4-NEXT:    [[TMP8:%.*]] = shl nuw i64 [[ADD5_7]], 55
 ; SSE4-NEXT:    [[RETVAL_SROA_8_0_INSERT_EXT:%.*]] = and i64 [[TMP8]], -72057594037927936
 ; SSE4-NEXT:    [[TMP9:%.*]] = shl nuw nsw i64 [[ADD5_6]], 47
-; SSE4-NEXT:    [[RETVAL_SROA_7_0_INSERT_SHIFT:%.*]] = and i64 [[TMP9]], 71776119061217280
+; SSE4-NEXT:    [[RETVAL_SROA_7_0_INSERT_SHIFT:%.*]] = and i64 [[TMP9]], 143833713099145216
 ; SSE4-NEXT:    [[TMP10:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_6_0_EXTRACT_SHIFT]], i64 0
 ; SSE4-NEXT:    [[TMP11:%.*]] = insertelement <4 x i64> [[TMP10]], i64 [[A_SROA_5_0_EXTRACT_SHIFT]], i64 1
 ; SSE4-NEXT:    [[TMP12:%.*]] = insertelement <4 x i64> [[TMP11]], i64 [[A_SROA_4_0_EXTRACT_SHIFT]], i64 2
@@ -174,18 +176,17 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE4-NEXT:    [[TMP24:%.*]] = bitcast <4 x i32> [[TMP23]] to <16 x i8>
 ; SSE4-NEXT:    [[TMP25:%.*]] = shufflevector <16 x i8> [[TMP24]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
 ; SSE4-NEXT:    [[TMP26:%.*]] = bitcast <8 x i8> [[TMP25]] to i64
-; SSE4-NEXT:    [[TMP27:%.*]] = shl nuw i16 [[ADD5_1]], 7
-; SSE4-NEXT:    [[TMP28:%.*]] = and i16 [[TMP27]], -256
-; SSE4-NEXT:    [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP28]] to i64
-; SSE4-NEXT:    [[OP_RDX5:%.*]] = or disjoint i64 [[RETVAL_SROA_8_0_INSERT_EXT]], [[TMP26]]
-; SSE4-NEXT:    [[OP_RDX6:%.*]] = or disjoint i64 [[RETVAL_SROA_7_0_INSERT_SHIFT]], [[SHR]]
-; SSE4-NEXT:    [[OP_RDX7:%.*]] = or disjoint i64 [[OP_RDX5]], [[OP_RDX6]]
+; SSE4-NEXT:    [[RETVAL_SROA_2_0_INSERT_EXT:%.*]] = zext nneg i16 [[SHR_1]] to i64
+; SSE4-NEXT:    [[RETVAL_SROA_2_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_2_0_INSERT_EXT]], 8
+; SSE4-NEXT:    [[OP_RDX7:%.*]] = or disjoint i64 [[RETVAL_SROA_8_0_INSERT_EXT]], [[TMP26]]
+; SSE4-NEXT:    [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED:%.*]] = or disjoint i64 [[RETVAL_SROA_7_0_INSERT_SHIFT]], [[SHR]]
 ; SSE4-NEXT:    [[TMP68:%.*]] = or i64 [[OP_RDX7]], [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED]]
-; SSE4-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP68]], 0
+; SSE4-NEXT:    [[OP_RDX8:%.*]] = or i64 [[TMP68]], [[RETVAL_SROA_2_0_INSERT_SHIFT]]
+; SSE4-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[OP_RDX8]], 0
 ; SSE4-NEXT:    [[TMP29:%.*]] = shl nuw i64 [[ADD5_15]], 55
 ; SSE4-NEXT:    [[RETVAL_SROA_17_8_INSERT_EXT:%.*]] = and i64 [[TMP29]], -72057594037927936
 ; SSE4-NEXT:    [[TMP30:%.*]] = shl nuw nsw i64 [[ADD5_14]], 47
-; SSE4-NEXT:    [[RETVAL_SROA_16_8_INSERT_SHIFT:%.*]] = and i64 [[TMP30]], 71776119061217280
+; SSE4-NEXT:    [[RETVAL_SROA_16_8_INSERT_SHIFT:%.*]] = and i64 [[TMP30]], 143833713099145216
 ; SSE4-NEXT:    [[TMP31:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_15_8_EXTRACT_SHIFT]], i64 0
 ; SSE4-NEXT:    [[TMP32:%.*]] = insertelement <4 x i64> [[TMP31]], i64 [[A_SROA_14_8_EXTRACT_SHIFT]], i64 1
 ; SSE4-NEXT:    [[TMP33:%.*]] = insertelement <4 x i64> [[TMP32]], i64 [[A_SROA_13_8_EXTRACT_SHIFT]], i64 2
@@ -203,14 +204,13 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE4-NEXT:    [[TMP45:%.*]] = bitcast <4 x i32> [[TMP44]] to <16 x i8>
 ; SSE4-NEXT:    [[TMP46:%.*]] = shufflevector <16 x i8> [[TMP45]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
 ; SSE4-NEXT:    [[TMP47:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
-; SSE4-NEXT:    [[TMP48:%.*]] = shl nuw i16 [[ADD5_9]], 7
-; SSE4-NEXT:    [[TMP49:%.*]] = and i16 [[TMP48]], -256
-; SSE4-NEXT:    [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP49]] to i64
-; SSE4-NEXT:    [[OP_RDX:%.*]] = or disjoint i64 [[RETVAL_SROA_17_8_INSERT_EXT]], [[TMP47]]
-; SSE4-NEXT:    [[OP_RDX2:%.*]] = or disjoint i64 [[RETVAL_SROA_16_8_INSERT_SHIFT]], [[SHR_8]]
-; SSE4-NEXT:    [[OP_RDX3:%.*]] = or disjoint i64 [[OP_RDX]], [[OP_RDX2]]
+; SSE4-NEXT:    [[RETVAL_SROA_11_8_INSERT_EXT:%.*]] = zext nneg i16 [[SHR_9]] to i64
+; SSE4-NEXT:    [[RETVAL_SROA_11_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_11_8_INSERT_EXT]], 8
+; SSE4-NEXT:    [[OP_RDX3:%.*]] = or disjoint i64 [[RETVAL_SROA_17_8_INSERT_EXT]], [[TMP47]]
+; SSE4-NEXT:    [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED:%.*]] = or disjoint i64 [[RETVAL_SROA_16_8_INSERT_SHIFT]], [[SHR_8]]
 ; SSE4-NEXT:    [[TMP69:%.*]] = or i64 [[OP_RDX3]], [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED]]
-; SSE4-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP69]], 1
+; SSE4-NEXT:    [[OP_RDX4:%.*]] = or i64 [[TMP69]], [[RETVAL_SROA_11_8_INSERT_SHIFT]]
+; SSE4-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[OP_RDX4]], 1
 ; SSE4-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
 ; AVX2-LABEL: @avgr_16_u8(
@@ -223,55 +223,44 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; AVX2-NEXT:    [[TMP5:%.*]] = insertelement <2 x i16> poison, i16 [[TMP4]], i64 0
 ; AVX2-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i16
 ; AVX2-NEXT:    [[TMP7:%.*]] = insertelement <2 x i16> [[TMP5]], i16 [[TMP6]], i64 1
-; AVX2-NEXT:    [[CONV1:%.*]] = and i64 [[A_COERCE0]], 255
-; AVX2-NEXT:    [[CONV4:%.*]] = and i64 [[B_COERCE0]], 255
-; AVX2-NEXT:    [[ADD:%.*]] = add nuw nsw i64 [[CONV1]], 1
-; AVX2-NEXT:    [[ADD5:%.*]] = add nuw nsw i64 [[ADD]], [[CONV4]]
-; AVX2-NEXT:    [[CONV1_8:%.*]] = and i64 [[A_COERCE1]], 255
-; AVX2-NEXT:    [[CONV4_8:%.*]] = and i64 [[B_COERCE1]], 255
-; AVX2-NEXT:    [[ADD_8:%.*]] = add nuw nsw i64 [[CONV1_8]], 1
-; AVX2-NEXT:    [[ADD5_8:%.*]] = add nuw nsw i64 [[ADD_8]], [[CONV4_8]]
-; AVX2-NEXT:    [[TMP8:%.*]] = insertelement <8 x i64> <i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 -1, i64 -1>, i64 [[B_COERCE0]], i64 0
-; AVX2-NEXT:    [[TMP9:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 6, i32 7>
+; AVX2-NEXT:    [[TMP8:%.*]] = insertelement <8 x i64> <i64 poison, i64 -1, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison>, i64 [[B_COERCE0]], i64 0
+; AVX2-NEXT:    [[TMP9:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1>
 ; AVX2-NEXT:    [[TMP10:%.*]] = lshr <8 x i64> [[TMP9]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
 ; AVX2-NEXT:    [[TMP11:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE0]], i64 0
-; AVX2-NEXT:    [[TMP12:%.*]] = insertelement <8 x i64> [[TMP11]...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/225206


More information about the llvm-commits mailing list