[llvm] [InstCombine] Optimize `zext(trunc nuw x)` (PR #225206)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 21 17:55:57 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
Author: peterbell10
<details>
<summary>Changes</summary>
We already have the equivalent transformation for `sext(trunc nsw x)`, this factors it into `commonCastTransforms` and generalizes it to work for `zext` with `nuw`.
Alive2 proofs for `zext`: https://alive2.llvm.org/ce/z/r-2bfQ
Alive2 proofs for `sext`: https://alive2.llvm.org/ce/z/VF6AYg
Assisted-by: GPT-6 Astra
---
Patch is 53.89 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/225206.diff
5 Files Affected:
- (modified) llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp (+19-7)
- (modified) llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll (+40)
- (modified) llvm/test/Transforms/InstCombine/trunc.ll (+2-4)
- (modified) llvm/test/Transforms/InstCombine/zext.ll (+104)
- (modified) llvm/test/Transforms/PhaseOrdering/X86/avg.ll (+263-146)
``````````diff
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
index 97defc5e3ddd3..274eed449dc59 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
@@ -239,6 +239,23 @@ Instruction *InstCombinerImpl::commonCastTransforms(CastInst &CI) {
replaceAllDbgUsesWith(*CSrc, *Res, CI, DT);
return Res;
}
+
+ // A no-wrap trunc followed by the corresponding extension preserves the
+ // value, so cast directly to the final type.
+ bool IsZExt = isa<ZExtInst>(CI);
+ bool IsSExt = isa<SExtInst>(CI);
+ if (auto *Trunc = dyn_cast<TruncInst>(CSrc);
+ Trunc && ((IsZExt && Trunc->hasNoUnsignedWrap()) ||
+ (IsSExt && Trunc->hasNoSignedWrap()))) {
+ auto *Res = CastInst::CreateIntegerCast(Trunc->getOperand(0), Ty, IsSExt);
+ if (auto *ResTrunc = dyn_cast<TruncInst>(Res)) {
+ ResTrunc->setHasNoUnsignedWrap(Trunc->hasNoUnsignedWrap());
+ ResTrunc->setHasNoSignedWrap(Trunc->hasNoSignedWrap());
+ } else if (auto *ResZExt = dyn_cast<ZExtInst>(Res)) {
+ ResZExt->setNonNeg(true);
+ }
+ return Res;
+ }
}
if (auto *Sel = dyn_cast<SelectInst>(Src)) {
@@ -1967,13 +1984,8 @@ Instruction *InstCombinerImpl::visitSExt(SExtInst &Sext) {
// If the input has more sign bits than bits truncated, then convert
// directly to final type.
unsigned XBitSize = X->getType()->getScalarSizeInBits();
- bool HasNSW = cast<TruncInst>(Src)->hasNoSignedWrap();
- if (HasNSW || (ComputeNumSignBits(X, &Sext) > XBitSize - SrcBitSize)) {
- auto *Res = CastInst::CreateIntegerCast(X, DestTy, /* isSigned */ true);
- if (auto *ResTrunc = dyn_cast<TruncInst>(Res); ResTrunc && HasNSW)
- ResTrunc->setHasNoSignedWrap(true);
- return Res;
- }
+ if (ComputeNumSignBits(X, &Sext) > XBitSize - SrcBitSize)
+ return CastInst::CreateIntegerCast(X, DestTy, /* isSigned */ true);
// If input is a trunc from the destination type, then convert into shifts.
if (Src->hasOneUse() && X->getType() == DestTy) {
diff --git a/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll b/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll
index 4128a15d8d7ce..d6fb3a10fccd6 100644
--- a/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll
+++ b/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll
@@ -233,3 +233,43 @@ define i32 @same_source_not_matching_signbits_extra_use(i32 %x) {
%c = sext i8 %b to i32
ret i32 %c
}
+
+define i32 @sext_trunc_widen(i8 %x) {
+; CHECK-LABEL: @sext_trunc_widen(
+; CHECK-NEXT: [[RESULT:%.*]] = sext i8 [[X:%.*]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc nsw i8 %x to i1
+ %result = sext i1 %narrow to i32
+ ret i32 %result
+}
+
+define i8 @sext_trunc_same(i8 %x) {
+; CHECK-LABEL: @sext_trunc_same(
+; CHECK-NEXT: ret i8 [[X:%.*]]
+;
+ %narrow = trunc nsw i8 %x to i1
+ %result = sext i1 %narrow to i8
+ ret i8 %result
+}
+
+define i32 @sext_trunc_narrow(i64 %x) {
+; CHECK-LABEL: @sext_trunc_narrow(
+; CHECK-NEXT: [[RESULT:%.*]] = trunc nsw i64 [[X:%.*]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc nsw i64 %x to i1
+ %result = sext i1 %narrow to i32
+ ret i32 %result
+}
+
+define i32 @sext_trunc_nuw(i8 %x) {
+; CHECK-LABEL: @sext_trunc_nuw(
+; CHECK-NEXT: [[NARROW:%.*]] = trunc nuw i8 [[X:%.*]] to i1
+; CHECK-NEXT: [[RESULT:%.*]] = sext i1 [[NARROW]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc nuw i8 %x to i1
+ %result = sext i1 %narrow to i32
+ ret i32 %result
+}
diff --git a/llvm/test/Transforms/InstCombine/trunc.ll b/llvm/test/Transforms/InstCombine/trunc.ll
index 91f637e89069f..6f99bf444ccb4 100644
--- a/llvm/test/Transforms/InstCombine/trunc.ll
+++ b/llvm/test/Transforms/InstCombine/trunc.ll
@@ -1406,8 +1406,7 @@ define i32 @neg_zext_i32_trunc_nsw_i8(i16 %x, i32 %y) {
define i16 @zext_i16_trunc_nuw_nsw_i8(i32 %x) {
; CHECK-LABEL: @zext_i16_trunc_nuw_nsw_i8(
-; CHECK-NEXT: [[C:%.*]] = trunc nuw nsw i32 [[X:%.*]] to i16
-; CHECK-NEXT: [[E:%.*]] = and i16 [[C]], 255
+; CHECK-NEXT: [[E:%.*]] = trunc nuw nsw i32 [[X:%.*]] to i16
; CHECK-NEXT: ret i16 [[E]]
;
%c = trunc nuw nsw i32 %x to i8
@@ -1428,8 +1427,7 @@ define i16 @zext_i16_trunc_nsw_i8(i32 %x) {
define i16 @zext_i16_trunc_nuw_i8(i32 %x) {
; CHECK-LABEL: @zext_i16_trunc_nuw_i8(
-; CHECK-NEXT: [[C:%.*]] = trunc nuw i32 [[X:%.*]] to i16
-; CHECK-NEXT: [[E:%.*]] = and i16 [[C]], 255
+; CHECK-NEXT: [[E:%.*]] = trunc nuw i32 [[X:%.*]] to i16
; CHECK-NEXT: ret i16 [[E]]
;
%c = trunc nuw i32 %x to i8
diff --git a/llvm/test/Transforms/InstCombine/zext.ll b/llvm/test/Transforms/InstCombine/zext.ll
index 3fb1e77ae2335..e98710d6b43d1 100644
--- a/llvm/test/Transforms/InstCombine/zext.ll
+++ b/llvm/test/Transforms/InstCombine/zext.ll
@@ -1080,3 +1080,107 @@ define <2 x i8> @zext_or_trunc_nuw_vec(<2 x i8> %x, <2 x i4> %y) {
%zext = zext <2 x i4> %or to <2 x i8>
ret <2 x i8> %zext
}
+
+define i32 @zext_trunc_widen(i8 %x) {
+; CHECK-LABEL: @zext_trunc_widen(
+; CHECK-NEXT: [[RESULT:%.*]] = zext nneg i8 [[X:%.*]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc nuw i8 %x to i1
+ %result = zext i1 %narrow to i32
+ ret i32 %result
+}
+
+define i64 @zext_trunc_native(i8 %x) {
+; CHECK-LABEL: @zext_trunc_native(
+; CHECK-NEXT: [[RESULT:%.*]] = zext nneg i8 [[X:%.*]] to i64
+; CHECK-NEXT: ret i64 [[RESULT]]
+;
+ %narrow = trunc nuw i8 %x to i1
+ %result = zext i1 %narrow to i64
+ ret i64 %result
+}
+
+define i8 @zext_trunc_same(i8 %x) {
+; CHECK-LABEL: @zext_trunc_same(
+; CHECK-NEXT: ret i8 [[X:%.*]]
+;
+ %narrow = trunc nuw i8 %x to i1
+ %result = zext i1 %narrow to i8
+ ret i8 %result
+}
+
+define i32 @zext_trunc_narrow(i64 %x) {
+; CHECK-LABEL: @zext_trunc_narrow(
+; CHECK-NEXT: [[RESULT:%.*]] = trunc nuw i64 [[X:%.*]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc nuw i64 %x to i1
+ %result = zext i1 %narrow to i32
+ ret i32 %result
+}
+
+define i32 @zext_trunc_no_flag(i8 %x) {
+; CHECK-LABEL: @zext_trunc_no_flag(
+; CHECK-NEXT: [[NARROW_MASK:%.*]] = and i8 [[X:%.*]], 1
+; CHECK-NEXT: [[RESULT:%.*]] = zext nneg i8 [[NARROW_MASK]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc i8 %x to i1
+ %result = zext i1 %narrow to i32
+ ret i32 %result
+}
+
+define i32 @zext_trunc_nsw(i8 %x) {
+; CHECK-LABEL: @zext_trunc_nsw(
+; CHECK-NEXT: [[NARROW_MASK:%.*]] = and i8 [[X:%.*]], 1
+; CHECK-NEXT: [[RESULT:%.*]] = zext nneg i8 [[NARROW_MASK]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc nsw i8 %x to i1
+ %result = zext i1 %narrow to i32
+ ret i32 %result
+}
+
+define <2 x i32> @zext_trunc_nuw_vec(<2 x i8> %x) {
+; CHECK-LABEL: @zext_trunc_nuw_vec(
+; CHECK-NEXT: [[RESULT:%.*]] = zext nneg <2 x i8> [[X:%.*]] to <2 x i32>
+; CHECK-NEXT: ret <2 x i32> [[RESULT]]
+;
+ %narrow = trunc nuw <2 x i8> %x to <2 x i1>
+ %result = zext <2 x i1> %narrow to <2 x i32>
+ ret <2 x i32> %result
+}
+
+define <vscale x 2 x i32> @zext_trunc_nuw_scalable(<vscale x 2 x i8> %x) {
+; CHECK-LABEL: @zext_trunc_nuw_scalable(
+; CHECK-NEXT: [[RESULT:%.*]] = zext nneg <vscale x 2 x i8> [[X:%.*]] to <vscale x 2 x i32>
+; CHECK-NEXT: ret <vscale x 2 x i32> [[RESULT]]
+;
+ %narrow = trunc nuw <vscale x 2 x i8> %x to <vscale x 2 x i1>
+ %result = zext <vscale x 2 x i1> %narrow to <vscale x 2 x i32>
+ ret <vscale x 2 x i32> %result
+}
+
+define <2 x i32> @zext_trunc_nuw_vec_narrow(<2 x i64> %x) {
+; CHECK-LABEL: @zext_trunc_nuw_vec_narrow(
+; CHECK-NEXT: [[RESULT:%.*]] = trunc nuw <2 x i64> [[X:%.*]] to <2 x i32>
+; CHECK-NEXT: ret <2 x i32> [[RESULT]]
+;
+ %narrow = trunc nuw <2 x i64> %x to <2 x i1>
+ %result = zext <2 x i1> %narrow to <2 x i32>
+ ret <2 x i32> %result
+}
+
+define i32 @zext_trunc_nuw_multi_use(i8 %x) {
+; CHECK-LABEL: @zext_trunc_nuw_multi_use(
+; CHECK-NEXT: [[NARROW:%.*]] = trunc nuw i8 [[X:%.*]] to i1
+; CHECK-NEXT: call void @use1(i1 [[NARROW]])
+; CHECK-NEXT: [[RESULT:%.*]] = zext nneg i8 [[X]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc nuw i8 %x to i1
+ call void @use1(i1 %narrow)
+ %result = zext i1 %narrow to i32
+ ret i32 %result
+}
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
index 8f8899d285ad3..7abb52200084a 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
@@ -44,6 +44,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; SSE2-NEXT: [[TMP26:%.*]] = lshr <2 x i64> [[TMP25]], splat (i64 1)
; SSE2-NEXT: [[TMP27:%.*]] = add nuw nsw <2 x i16> [[TMP20]], splat (i16 1)
; SSE2-NEXT: [[TMP28:%.*]] = add nuw nsw <2 x i16> [[TMP27]], [[TMP23]]
+; SSE2-NEXT: [[TMP50:%.*]] = lshr <2 x i16> [[TMP28]], splat (i16 1)
; SSE2-NEXT: [[TMP29:%.*]] = and <2 x i64> [[TMP3]], splat (i64 255)
; SSE2-NEXT: [[TMP30:%.*]] = and <2 x i64> [[TMP11]], splat (i64 255)
; SSE2-NEXT: [[TMP31:%.*]] = add nuw nsw <2 x i64> [[TMP29]], splat (i64 1)
@@ -68,25 +69,24 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; SSE2-NEXT: [[TMP75:%.*]] = shl nuw <2 x i64> [[TMP74]], splat (i64 55)
; SSE2-NEXT: [[TMP76:%.*]] = and <2 x i64> [[TMP75]], splat (i64 -72057594037927936)
; SSE2-NEXT: [[TMP77:%.*]] = shl nuw nsw <2 x i64> [[TMP72]], splat (i64 47)
-; SSE2-NEXT: [[TMP78:%.*]] = and <2 x i64> [[TMP77]], splat (i64 71776119061217280)
-; SSE2-NEXT: [[TMP52:%.*]] = or disjoint <2 x i64> [[TMP76]], [[TMP78]]
+; SSE2-NEXT: [[TMP53:%.*]] = and <2 x i64> [[TMP77]], splat (i64 143833713099145216)
+; SSE2-NEXT: [[TMP55:%.*]] = or <2 x i64> [[TMP76]], [[TMP53]]
; SSE2-NEXT: [[TMP49:%.*]] = shl nuw nsw <2 x i64> [[TMP44]], splat (i64 39)
-; SSE2-NEXT: [[TMP50:%.*]] = and <2 x i64> [[TMP49]], splat (i64 280375465082880)
-; SSE2-NEXT: [[TMP53:%.*]] = or disjoint <2 x i64> [[TMP52]], [[TMP50]]
+; SSE2-NEXT: [[TMP56:%.*]] = and <2 x i64> [[TMP49]], splat (i64 561850441793536)
+; SSE2-NEXT: [[TMP58:%.*]] = or <2 x i64> [[TMP55]], [[TMP56]]
; SSE2-NEXT: [[TMP54:%.*]] = shl nuw nsw <2 x i64> [[TMP40]], splat (i64 31)
-; SSE2-NEXT: [[TMP55:%.*]] = and <2 x i64> [[TMP54]], splat (i64 1095216660480)
-; SSE2-NEXT: [[TMP56:%.*]] = or disjoint <2 x i64> [[TMP53]], [[TMP55]]
+; SSE2-NEXT: [[TMP59:%.*]] = and <2 x i64> [[TMP54]], splat (i64 2194728288256)
+; SSE2-NEXT: [[TMP61:%.*]] = or <2 x i64> [[TMP58]], [[TMP59]]
; SSE2-NEXT: [[TMP57:%.*]] = shl nuw nsw <2 x i64> [[TMP36]], splat (i64 23)
-; SSE2-NEXT: [[TMP58:%.*]] = and <2 x i64> [[TMP57]], splat (i64 4278190080)
-; SSE2-NEXT: [[TMP59:%.*]] = or disjoint <2 x i64> [[TMP56]], [[TMP58]]
+; SSE2-NEXT: [[TMP62:%.*]] = and <2 x i64> [[TMP57]], splat (i64 8573157376)
+; SSE2-NEXT: [[TMP63:%.*]] = or <2 x i64> [[TMP61]], [[TMP62]]
; SSE2-NEXT: [[TMP60:%.*]] = shl nuw nsw <2 x i64> [[TMP32]], splat (i64 15)
-; SSE2-NEXT: [[TMP61:%.*]] = and <2 x i64> [[TMP60]], splat (i64 16711680)
-; SSE2-NEXT: [[TMP62:%.*]] = shl nuw <2 x i16> [[TMP28]], splat (i16 7)
-; SSE2-NEXT: [[TMP63:%.*]] = or disjoint <2 x i64> [[TMP59]], [[TMP61]]
-; SSE2-NEXT: [[TMP64:%.*]] = and <2 x i16> [[TMP62]], splat (i16 -256)
-; SSE2-NEXT: [[TMP65:%.*]] = zext <2 x i16> [[TMP64]] to <2 x i64>
+; SSE2-NEXT: [[TMP65:%.*]] = and <2 x i64> [[TMP60]], splat (i64 33488896)
+; SSE2-NEXT: [[TMP78:%.*]] = zext nneg <2 x i16> [[TMP50]] to <2 x i64>
+; SSE2-NEXT: [[TMP79:%.*]] = shl nuw nsw <2 x i64> [[TMP78]], splat (i64 8)
; SSE2-NEXT: [[TMP66:%.*]] = or <2 x i64> [[TMP63]], [[TMP65]]
-; SSE2-NEXT: [[TMP67:%.*]] = or <2 x i64> [[TMP66]], [[TMP26]]
+; SSE2-NEXT: [[TMP80:%.*]] = or <2 x i64> [[TMP66]], [[TMP79]]
+; SSE2-NEXT: [[TMP67:%.*]] = or <2 x i64> [[TMP80]], [[TMP26]]
; SSE2-NEXT: [[TMP68:%.*]] = extractelement <2 x i64> [[TMP67]], i64 0
; SSE2-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP68]], 0
; SSE2-NEXT: [[TMP69:%.*]] = extractelement <2 x i64> [[TMP67]], i64 1
@@ -134,6 +134,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; SSE4-NEXT: [[SHR:%.*]] = lshr i64 [[ADD5]], 1
; SSE4-NEXT: [[ADD_1:%.*]] = add nuw nsw i16 [[TMP1]], 1
; SSE4-NEXT: [[ADD5_1:%.*]] = add nuw nsw i16 [[ADD_1]], [[TMP5]]
+; SSE4-NEXT: [[SHR_1:%.*]] = lshr i16 [[ADD5_1]], 1
; SSE4-NEXT: [[CONV1_6:%.*]] = and i64 [[A_SROA_7_0_EXTRACT_SHIFT]], 255
; SSE4-NEXT: [[CONV4_6:%.*]] = and i64 [[B_SROA_7_0_EXTRACT_SHIFT]], 255
; SSE4-NEXT: [[ADD_6:%.*]] = add nuw nsw i64 [[CONV1_6]], 1
@@ -147,6 +148,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; SSE4-NEXT: [[SHR_8:%.*]] = lshr i64 [[ADD5_8]], 1
; SSE4-NEXT: [[ADD_9:%.*]] = add nuw nsw i16 [[TMP3]], 1
; SSE4-NEXT: [[ADD5_9:%.*]] = add nuw nsw i16 [[ADD_9]], [[TMP7]]
+; SSE4-NEXT: [[SHR_9:%.*]] = lshr i16 [[ADD5_9]], 1
; SSE4-NEXT: [[CONV1_14:%.*]] = and i64 [[A_SROA_16_8_EXTRACT_SHIFT]], 255
; SSE4-NEXT: [[CONV4_14:%.*]] = and i64 [[B_SROA_16_8_EXTRACT_SHIFT]], 255
; SSE4-NEXT: [[ADD_14:%.*]] = add nuw nsw i64 [[CONV1_14]], 1
@@ -156,7 +158,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; SSE4-NEXT: [[TMP8:%.*]] = shl nuw i64 [[ADD5_7]], 55
; SSE4-NEXT: [[RETVAL_SROA_8_0_INSERT_EXT:%.*]] = and i64 [[TMP8]], -72057594037927936
; SSE4-NEXT: [[TMP9:%.*]] = shl nuw nsw i64 [[ADD5_6]], 47
-; SSE4-NEXT: [[RETVAL_SROA_7_0_INSERT_SHIFT:%.*]] = and i64 [[TMP9]], 71776119061217280
+; SSE4-NEXT: [[RETVAL_SROA_7_0_INSERT_SHIFT:%.*]] = and i64 [[TMP9]], 143833713099145216
; SSE4-NEXT: [[TMP10:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_6_0_EXTRACT_SHIFT]], i64 0
; SSE4-NEXT: [[TMP11:%.*]] = insertelement <4 x i64> [[TMP10]], i64 [[A_SROA_5_0_EXTRACT_SHIFT]], i64 1
; SSE4-NEXT: [[TMP12:%.*]] = insertelement <4 x i64> [[TMP11]], i64 [[A_SROA_4_0_EXTRACT_SHIFT]], i64 2
@@ -174,18 +176,17 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; SSE4-NEXT: [[TMP24:%.*]] = bitcast <4 x i32> [[TMP23]] to <16 x i8>
; SSE4-NEXT: [[TMP25:%.*]] = shufflevector <16 x i8> [[TMP24]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
; SSE4-NEXT: [[TMP26:%.*]] = bitcast <8 x i8> [[TMP25]] to i64
-; SSE4-NEXT: [[TMP27:%.*]] = shl nuw i16 [[ADD5_1]], 7
-; SSE4-NEXT: [[TMP28:%.*]] = and i16 [[TMP27]], -256
-; SSE4-NEXT: [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP28]] to i64
-; SSE4-NEXT: [[OP_RDX5:%.*]] = or disjoint i64 [[RETVAL_SROA_8_0_INSERT_EXT]], [[TMP26]]
-; SSE4-NEXT: [[OP_RDX6:%.*]] = or disjoint i64 [[RETVAL_SROA_7_0_INSERT_SHIFT]], [[SHR]]
-; SSE4-NEXT: [[OP_RDX7:%.*]] = or disjoint i64 [[OP_RDX5]], [[OP_RDX6]]
+; SSE4-NEXT: [[RETVAL_SROA_2_0_INSERT_EXT:%.*]] = zext nneg i16 [[SHR_1]] to i64
+; SSE4-NEXT: [[RETVAL_SROA_2_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_2_0_INSERT_EXT]], 8
+; SSE4-NEXT: [[OP_RDX7:%.*]] = or disjoint i64 [[RETVAL_SROA_8_0_INSERT_EXT]], [[TMP26]]
+; SSE4-NEXT: [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED:%.*]] = or disjoint i64 [[RETVAL_SROA_7_0_INSERT_SHIFT]], [[SHR]]
; SSE4-NEXT: [[TMP68:%.*]] = or i64 [[OP_RDX7]], [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED]]
-; SSE4-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP68]], 0
+; SSE4-NEXT: [[OP_RDX8:%.*]] = or i64 [[TMP68]], [[RETVAL_SROA_2_0_INSERT_SHIFT]]
+; SSE4-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[OP_RDX8]], 0
; SSE4-NEXT: [[TMP29:%.*]] = shl nuw i64 [[ADD5_15]], 55
; SSE4-NEXT: [[RETVAL_SROA_17_8_INSERT_EXT:%.*]] = and i64 [[TMP29]], -72057594037927936
; SSE4-NEXT: [[TMP30:%.*]] = shl nuw nsw i64 [[ADD5_14]], 47
-; SSE4-NEXT: [[RETVAL_SROA_16_8_INSERT_SHIFT:%.*]] = and i64 [[TMP30]], 71776119061217280
+; SSE4-NEXT: [[RETVAL_SROA_16_8_INSERT_SHIFT:%.*]] = and i64 [[TMP30]], 143833713099145216
; SSE4-NEXT: [[TMP31:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_15_8_EXTRACT_SHIFT]], i64 0
; SSE4-NEXT: [[TMP32:%.*]] = insertelement <4 x i64> [[TMP31]], i64 [[A_SROA_14_8_EXTRACT_SHIFT]], i64 1
; SSE4-NEXT: [[TMP33:%.*]] = insertelement <4 x i64> [[TMP32]], i64 [[A_SROA_13_8_EXTRACT_SHIFT]], i64 2
@@ -203,14 +204,13 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; SSE4-NEXT: [[TMP45:%.*]] = bitcast <4 x i32> [[TMP44]] to <16 x i8>
; SSE4-NEXT: [[TMP46:%.*]] = shufflevector <16 x i8> [[TMP45]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
; SSE4-NEXT: [[TMP47:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
-; SSE4-NEXT: [[TMP48:%.*]] = shl nuw i16 [[ADD5_9]], 7
-; SSE4-NEXT: [[TMP49:%.*]] = and i16 [[TMP48]], -256
-; SSE4-NEXT: [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP49]] to i64
-; SSE4-NEXT: [[OP_RDX:%.*]] = or disjoint i64 [[RETVAL_SROA_17_8_INSERT_EXT]], [[TMP47]]
-; SSE4-NEXT: [[OP_RDX2:%.*]] = or disjoint i64 [[RETVAL_SROA_16_8_INSERT_SHIFT]], [[SHR_8]]
-; SSE4-NEXT: [[OP_RDX3:%.*]] = or disjoint i64 [[OP_RDX]], [[OP_RDX2]]
+; SSE4-NEXT: [[RETVAL_SROA_11_8_INSERT_EXT:%.*]] = zext nneg i16 [[SHR_9]] to i64
+; SSE4-NEXT: [[RETVAL_SROA_11_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_11_8_INSERT_EXT]], 8
+; SSE4-NEXT: [[OP_RDX3:%.*]] = or disjoint i64 [[RETVAL_SROA_17_8_INSERT_EXT]], [[TMP47]]
+; SSE4-NEXT: [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED:%.*]] = or disjoint i64 [[RETVAL_SROA_16_8_INSERT_SHIFT]], [[SHR_8]]
; SSE4-NEXT: [[TMP69:%.*]] = or i64 [[OP_RDX3]], [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED]]
-; SSE4-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP69]], 1
+; SSE4-NEXT: [[OP_RDX4:%.*]] = or i64 [[TMP69]], [[RETVAL_SROA_11_8_INSERT_SHIFT]]
+; SSE4-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[OP_RDX4]], 1
; SSE4-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
; AVX2-LABEL: @avgr_16_u8(
@@ -223,55 +223,44 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; AVX2-NEXT: [[TMP5:%.*]] = insertelement <2 x i16> poison, i16 [[TMP4]], i64 0
; AVX2-NEXT: [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i16
; AVX2-NEXT: [[TMP7:%.*]] = insertelement <2 x i16> [[TMP5]], i16 [[TMP6]], i64 1
-; AVX2-NEXT: [[CONV1:%.*]] = and i64 [[A_COERCE0]], 255
-; AVX2-NEXT: [[CONV4:%.*]] = and i64 [[B_COERCE0]], 255
-; AVX2-NEXT: [[ADD:%.*]] = add nuw nsw i64 [[CONV1]], 1
-; AVX2-NEXT: [[ADD5:%.*]] = add nuw nsw i64 [[ADD]], [[CONV4]]
-; AVX2-NEXT: [[CONV1_8:%.*]] = and i64 [[A_COERCE1]], 255
-; AVX2-NEXT: [[CONV4_8:%.*]] = and i64 [[B_COERCE1]], 255
-; AVX2-NEXT: [[ADD_8:%.*]] = add nuw nsw i64 [[CONV1_8]], 1
-; AVX2-NEXT: [[ADD5_8:%.*]] = add nuw nsw i64 [[ADD_8]], [[CONV4_8]]
-; AVX2-NEXT: [[TMP8:%.*]] = insertelement <8 x i64> <i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 -1, i64 -1>, i64 [[B_COERCE0]], i64 0
-; AVX2-NEXT: [[TMP9:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 6, i32 7>
+; AVX2-NEXT: [[TMP8:%.*]] = insertelement <8 x i64> <i64 poison, i64 -1, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison>, i64 [[B_COERCE0]], i64 0
+; AVX2-NEXT: [[TMP9:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1>
; AVX2-NEXT: [[TMP10:%.*]] = lshr <8 x i64> [[TMP9]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
; AVX2-NEXT: [[TMP11:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE0]], i64 0
-; AVX2-NEXT: [[TMP12:%.*]] = insertelement <8 x i64> [[TMP11]...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/225206
More information about the llvm-commits
mailing list