[llvm] [InstCombine] Optimize `zext(trunc nuw x)` (PR #225206)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 21 16:53:33 PDT 2026
https://github.com/peterbell10 updated https://github.com/llvm/llvm-project/pull/225206
>From 0172007d4587facadd271cb459e0fda99da0a2e9 Mon Sep 17 00:00:00 2001
From: Peter Bell <peterbell10 at openai.com>
Date: Mon, 21 Sep 2026 22:38:21 +0100
Subject: [PATCH 1/2] [InstCombine] Fold zext(trunc nuw x)
---
.../InstCombine/InstCombineCasts.cpp | 26 +++--
.../InstCombine/sext-of-trunc-nsw.ll | 40 +++++++
llvm/test/Transforms/InstCombine/trunc.ll | 6 +-
llvm/test/Transforms/InstCombine/zext.ll | 104 ++++++++++++++++++
4 files changed, 165 insertions(+), 11 deletions(-)
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
index 97defc5e3ddd3..274eed449dc59 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
@@ -239,6 +239,23 @@ Instruction *InstCombinerImpl::commonCastTransforms(CastInst &CI) {
replaceAllDbgUsesWith(*CSrc, *Res, CI, DT);
return Res;
}
+
+ // A no-wrap trunc followed by the corresponding extension preserves the
+ // value, so cast directly to the final type.
+ bool IsZExt = isa<ZExtInst>(CI);
+ bool IsSExt = isa<SExtInst>(CI);
+ if (auto *Trunc = dyn_cast<TruncInst>(CSrc);
+ Trunc && ((IsZExt && Trunc->hasNoUnsignedWrap()) ||
+ (IsSExt && Trunc->hasNoSignedWrap()))) {
+ auto *Res = CastInst::CreateIntegerCast(Trunc->getOperand(0), Ty, IsSExt);
+ if (auto *ResTrunc = dyn_cast<TruncInst>(Res)) {
+ ResTrunc->setHasNoUnsignedWrap(Trunc->hasNoUnsignedWrap());
+ ResTrunc->setHasNoSignedWrap(Trunc->hasNoSignedWrap());
+ } else if (auto *ResZExt = dyn_cast<ZExtInst>(Res)) {
+ ResZExt->setNonNeg(true);
+ }
+ return Res;
+ }
}
if (auto *Sel = dyn_cast<SelectInst>(Src)) {
@@ -1967,13 +1984,8 @@ Instruction *InstCombinerImpl::visitSExt(SExtInst &Sext) {
// If the input has more sign bits than bits truncated, then convert
// directly to final type.
unsigned XBitSize = X->getType()->getScalarSizeInBits();
- bool HasNSW = cast<TruncInst>(Src)->hasNoSignedWrap();
- if (HasNSW || (ComputeNumSignBits(X, &Sext) > XBitSize - SrcBitSize)) {
- auto *Res = CastInst::CreateIntegerCast(X, DestTy, /* isSigned */ true);
- if (auto *ResTrunc = dyn_cast<TruncInst>(Res); ResTrunc && HasNSW)
- ResTrunc->setHasNoSignedWrap(true);
- return Res;
- }
+ if (ComputeNumSignBits(X, &Sext) > XBitSize - SrcBitSize)
+ return CastInst::CreateIntegerCast(X, DestTy, /* isSigned */ true);
// If input is a trunc from the destination type, then convert into shifts.
if (Src->hasOneUse() && X->getType() == DestTy) {
diff --git a/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll b/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll
index 4128a15d8d7ce..d6fb3a10fccd6 100644
--- a/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll
+++ b/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll
@@ -233,3 +233,43 @@ define i32 @same_source_not_matching_signbits_extra_use(i32 %x) {
%c = sext i8 %b to i32
ret i32 %c
}
+
+define i32 @sext_trunc_widen(i8 %x) {
+; CHECK-LABEL: @sext_trunc_widen(
+; CHECK-NEXT: [[RESULT:%.*]] = sext i8 [[X:%.*]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc nsw i8 %x to i1
+ %result = sext i1 %narrow to i32
+ ret i32 %result
+}
+
+define i8 @sext_trunc_same(i8 %x) {
+; CHECK-LABEL: @sext_trunc_same(
+; CHECK-NEXT: ret i8 [[X:%.*]]
+;
+ %narrow = trunc nsw i8 %x to i1
+ %result = sext i1 %narrow to i8
+ ret i8 %result
+}
+
+define i32 @sext_trunc_narrow(i64 %x) {
+; CHECK-LABEL: @sext_trunc_narrow(
+; CHECK-NEXT: [[RESULT:%.*]] = trunc nsw i64 [[X:%.*]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc nsw i64 %x to i1
+ %result = sext i1 %narrow to i32
+ ret i32 %result
+}
+
+define i32 @sext_trunc_nuw(i8 %x) {
+; CHECK-LABEL: @sext_trunc_nuw(
+; CHECK-NEXT: [[NARROW:%.*]] = trunc nuw i8 [[X:%.*]] to i1
+; CHECK-NEXT: [[RESULT:%.*]] = sext i1 [[NARROW]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc nuw i8 %x to i1
+ %result = sext i1 %narrow to i32
+ ret i32 %result
+}
diff --git a/llvm/test/Transforms/InstCombine/trunc.ll b/llvm/test/Transforms/InstCombine/trunc.ll
index 91f637e89069f..6f99bf444ccb4 100644
--- a/llvm/test/Transforms/InstCombine/trunc.ll
+++ b/llvm/test/Transforms/InstCombine/trunc.ll
@@ -1406,8 +1406,7 @@ define i32 @neg_zext_i32_trunc_nsw_i8(i16 %x, i32 %y) {
define i16 @zext_i16_trunc_nuw_nsw_i8(i32 %x) {
; CHECK-LABEL: @zext_i16_trunc_nuw_nsw_i8(
-; CHECK-NEXT: [[C:%.*]] = trunc nuw nsw i32 [[X:%.*]] to i16
-; CHECK-NEXT: [[E:%.*]] = and i16 [[C]], 255
+; CHECK-NEXT: [[E:%.*]] = trunc nuw nsw i32 [[X:%.*]] to i16
; CHECK-NEXT: ret i16 [[E]]
;
%c = trunc nuw nsw i32 %x to i8
@@ -1428,8 +1427,7 @@ define i16 @zext_i16_trunc_nsw_i8(i32 %x) {
define i16 @zext_i16_trunc_nuw_i8(i32 %x) {
; CHECK-LABEL: @zext_i16_trunc_nuw_i8(
-; CHECK-NEXT: [[C:%.*]] = trunc nuw i32 [[X:%.*]] to i16
-; CHECK-NEXT: [[E:%.*]] = and i16 [[C]], 255
+; CHECK-NEXT: [[E:%.*]] = trunc nuw i32 [[X:%.*]] to i16
; CHECK-NEXT: ret i16 [[E]]
;
%c = trunc nuw i32 %x to i8
diff --git a/llvm/test/Transforms/InstCombine/zext.ll b/llvm/test/Transforms/InstCombine/zext.ll
index 3fb1e77ae2335..e98710d6b43d1 100644
--- a/llvm/test/Transforms/InstCombine/zext.ll
+++ b/llvm/test/Transforms/InstCombine/zext.ll
@@ -1080,3 +1080,107 @@ define <2 x i8> @zext_or_trunc_nuw_vec(<2 x i8> %x, <2 x i4> %y) {
%zext = zext <2 x i4> %or to <2 x i8>
ret <2 x i8> %zext
}
+
+define i32 @zext_trunc_widen(i8 %x) {
+; CHECK-LABEL: @zext_trunc_widen(
+; CHECK-NEXT: [[RESULT:%.*]] = zext nneg i8 [[X:%.*]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc nuw i8 %x to i1
+ %result = zext i1 %narrow to i32
+ ret i32 %result
+}
+
+define i64 @zext_trunc_native(i8 %x) {
+; CHECK-LABEL: @zext_trunc_native(
+; CHECK-NEXT: [[RESULT:%.*]] = zext nneg i8 [[X:%.*]] to i64
+; CHECK-NEXT: ret i64 [[RESULT]]
+;
+ %narrow = trunc nuw i8 %x to i1
+ %result = zext i1 %narrow to i64
+ ret i64 %result
+}
+
+define i8 @zext_trunc_same(i8 %x) {
+; CHECK-LABEL: @zext_trunc_same(
+; CHECK-NEXT: ret i8 [[X:%.*]]
+;
+ %narrow = trunc nuw i8 %x to i1
+ %result = zext i1 %narrow to i8
+ ret i8 %result
+}
+
+define i32 @zext_trunc_narrow(i64 %x) {
+; CHECK-LABEL: @zext_trunc_narrow(
+; CHECK-NEXT: [[RESULT:%.*]] = trunc nuw i64 [[X:%.*]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc nuw i64 %x to i1
+ %result = zext i1 %narrow to i32
+ ret i32 %result
+}
+
+define i32 @zext_trunc_no_flag(i8 %x) {
+; CHECK-LABEL: @zext_trunc_no_flag(
+; CHECK-NEXT: [[NARROW_MASK:%.*]] = and i8 [[X:%.*]], 1
+; CHECK-NEXT: [[RESULT:%.*]] = zext nneg i8 [[NARROW_MASK]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc i8 %x to i1
+ %result = zext i1 %narrow to i32
+ ret i32 %result
+}
+
+define i32 @zext_trunc_nsw(i8 %x) {
+; CHECK-LABEL: @zext_trunc_nsw(
+; CHECK-NEXT: [[NARROW_MASK:%.*]] = and i8 [[X:%.*]], 1
+; CHECK-NEXT: [[RESULT:%.*]] = zext nneg i8 [[NARROW_MASK]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc nsw i8 %x to i1
+ %result = zext i1 %narrow to i32
+ ret i32 %result
+}
+
+define <2 x i32> @zext_trunc_nuw_vec(<2 x i8> %x) {
+; CHECK-LABEL: @zext_trunc_nuw_vec(
+; CHECK-NEXT: [[RESULT:%.*]] = zext nneg <2 x i8> [[X:%.*]] to <2 x i32>
+; CHECK-NEXT: ret <2 x i32> [[RESULT]]
+;
+ %narrow = trunc nuw <2 x i8> %x to <2 x i1>
+ %result = zext <2 x i1> %narrow to <2 x i32>
+ ret <2 x i32> %result
+}
+
+define <vscale x 2 x i32> @zext_trunc_nuw_scalable(<vscale x 2 x i8> %x) {
+; CHECK-LABEL: @zext_trunc_nuw_scalable(
+; CHECK-NEXT: [[RESULT:%.*]] = zext nneg <vscale x 2 x i8> [[X:%.*]] to <vscale x 2 x i32>
+; CHECK-NEXT: ret <vscale x 2 x i32> [[RESULT]]
+;
+ %narrow = trunc nuw <vscale x 2 x i8> %x to <vscale x 2 x i1>
+ %result = zext <vscale x 2 x i1> %narrow to <vscale x 2 x i32>
+ ret <vscale x 2 x i32> %result
+}
+
+define <2 x i32> @zext_trunc_nuw_vec_narrow(<2 x i64> %x) {
+; CHECK-LABEL: @zext_trunc_nuw_vec_narrow(
+; CHECK-NEXT: [[RESULT:%.*]] = trunc nuw <2 x i64> [[X:%.*]] to <2 x i32>
+; CHECK-NEXT: ret <2 x i32> [[RESULT]]
+;
+ %narrow = trunc nuw <2 x i64> %x to <2 x i1>
+ %result = zext <2 x i1> %narrow to <2 x i32>
+ ret <2 x i32> %result
+}
+
+define i32 @zext_trunc_nuw_multi_use(i8 %x) {
+; CHECK-LABEL: @zext_trunc_nuw_multi_use(
+; CHECK-NEXT: [[NARROW:%.*]] = trunc nuw i8 [[X:%.*]] to i1
+; CHECK-NEXT: call void @use1(i1 [[NARROW]])
+; CHECK-NEXT: [[RESULT:%.*]] = zext nneg i8 [[X]] to i32
+; CHECK-NEXT: ret i32 [[RESULT]]
+;
+ %narrow = trunc nuw i8 %x to i1
+ call void @use1(i1 %narrow)
+ %result = zext i1 %narrow to i32
+ ret i32 %result
+}
>From c00534f56e70f66572d8de1ffc79eefcaa731ad6 Mon Sep 17 00:00:00 2001
From: Peter Bell <peterbell10 at openai.com>
Date: Tue, 22 Sep 2026 00:52:35 +0100
Subject: [PATCH 2/2] Fix x86 lit test
---
llvm/test/Transforms/PhaseOrdering/X86/avg.ll | 409 +++++++++++-------
1 file changed, 263 insertions(+), 146 deletions(-)
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
index 8f8899d285ad3..7abb52200084a 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
@@ -44,6 +44,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; SSE2-NEXT: [[TMP26:%.*]] = lshr <2 x i64> [[TMP25]], splat (i64 1)
; SSE2-NEXT: [[TMP27:%.*]] = add nuw nsw <2 x i16> [[TMP20]], splat (i16 1)
; SSE2-NEXT: [[TMP28:%.*]] = add nuw nsw <2 x i16> [[TMP27]], [[TMP23]]
+; SSE2-NEXT: [[TMP50:%.*]] = lshr <2 x i16> [[TMP28]], splat (i16 1)
; SSE2-NEXT: [[TMP29:%.*]] = and <2 x i64> [[TMP3]], splat (i64 255)
; SSE2-NEXT: [[TMP30:%.*]] = and <2 x i64> [[TMP11]], splat (i64 255)
; SSE2-NEXT: [[TMP31:%.*]] = add nuw nsw <2 x i64> [[TMP29]], splat (i64 1)
@@ -68,25 +69,24 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; SSE2-NEXT: [[TMP75:%.*]] = shl nuw <2 x i64> [[TMP74]], splat (i64 55)
; SSE2-NEXT: [[TMP76:%.*]] = and <2 x i64> [[TMP75]], splat (i64 -72057594037927936)
; SSE2-NEXT: [[TMP77:%.*]] = shl nuw nsw <2 x i64> [[TMP72]], splat (i64 47)
-; SSE2-NEXT: [[TMP78:%.*]] = and <2 x i64> [[TMP77]], splat (i64 71776119061217280)
-; SSE2-NEXT: [[TMP52:%.*]] = or disjoint <2 x i64> [[TMP76]], [[TMP78]]
+; SSE2-NEXT: [[TMP53:%.*]] = and <2 x i64> [[TMP77]], splat (i64 143833713099145216)
+; SSE2-NEXT: [[TMP55:%.*]] = or <2 x i64> [[TMP76]], [[TMP53]]
; SSE2-NEXT: [[TMP49:%.*]] = shl nuw nsw <2 x i64> [[TMP44]], splat (i64 39)
-; SSE2-NEXT: [[TMP50:%.*]] = and <2 x i64> [[TMP49]], splat (i64 280375465082880)
-; SSE2-NEXT: [[TMP53:%.*]] = or disjoint <2 x i64> [[TMP52]], [[TMP50]]
+; SSE2-NEXT: [[TMP56:%.*]] = and <2 x i64> [[TMP49]], splat (i64 561850441793536)
+; SSE2-NEXT: [[TMP58:%.*]] = or <2 x i64> [[TMP55]], [[TMP56]]
; SSE2-NEXT: [[TMP54:%.*]] = shl nuw nsw <2 x i64> [[TMP40]], splat (i64 31)
-; SSE2-NEXT: [[TMP55:%.*]] = and <2 x i64> [[TMP54]], splat (i64 1095216660480)
-; SSE2-NEXT: [[TMP56:%.*]] = or disjoint <2 x i64> [[TMP53]], [[TMP55]]
+; SSE2-NEXT: [[TMP59:%.*]] = and <2 x i64> [[TMP54]], splat (i64 2194728288256)
+; SSE2-NEXT: [[TMP61:%.*]] = or <2 x i64> [[TMP58]], [[TMP59]]
; SSE2-NEXT: [[TMP57:%.*]] = shl nuw nsw <2 x i64> [[TMP36]], splat (i64 23)
-; SSE2-NEXT: [[TMP58:%.*]] = and <2 x i64> [[TMP57]], splat (i64 4278190080)
-; SSE2-NEXT: [[TMP59:%.*]] = or disjoint <2 x i64> [[TMP56]], [[TMP58]]
+; SSE2-NEXT: [[TMP62:%.*]] = and <2 x i64> [[TMP57]], splat (i64 8573157376)
+; SSE2-NEXT: [[TMP63:%.*]] = or <2 x i64> [[TMP61]], [[TMP62]]
; SSE2-NEXT: [[TMP60:%.*]] = shl nuw nsw <2 x i64> [[TMP32]], splat (i64 15)
-; SSE2-NEXT: [[TMP61:%.*]] = and <2 x i64> [[TMP60]], splat (i64 16711680)
-; SSE2-NEXT: [[TMP62:%.*]] = shl nuw <2 x i16> [[TMP28]], splat (i16 7)
-; SSE2-NEXT: [[TMP63:%.*]] = or disjoint <2 x i64> [[TMP59]], [[TMP61]]
-; SSE2-NEXT: [[TMP64:%.*]] = and <2 x i16> [[TMP62]], splat (i16 -256)
-; SSE2-NEXT: [[TMP65:%.*]] = zext <2 x i16> [[TMP64]] to <2 x i64>
+; SSE2-NEXT: [[TMP65:%.*]] = and <2 x i64> [[TMP60]], splat (i64 33488896)
+; SSE2-NEXT: [[TMP78:%.*]] = zext nneg <2 x i16> [[TMP50]] to <2 x i64>
+; SSE2-NEXT: [[TMP79:%.*]] = shl nuw nsw <2 x i64> [[TMP78]], splat (i64 8)
; SSE2-NEXT: [[TMP66:%.*]] = or <2 x i64> [[TMP63]], [[TMP65]]
-; SSE2-NEXT: [[TMP67:%.*]] = or <2 x i64> [[TMP66]], [[TMP26]]
+; SSE2-NEXT: [[TMP80:%.*]] = or <2 x i64> [[TMP66]], [[TMP79]]
+; SSE2-NEXT: [[TMP67:%.*]] = or <2 x i64> [[TMP80]], [[TMP26]]
; SSE2-NEXT: [[TMP68:%.*]] = extractelement <2 x i64> [[TMP67]], i64 0
; SSE2-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP68]], 0
; SSE2-NEXT: [[TMP69:%.*]] = extractelement <2 x i64> [[TMP67]], i64 1
@@ -134,6 +134,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; SSE4-NEXT: [[SHR:%.*]] = lshr i64 [[ADD5]], 1
; SSE4-NEXT: [[ADD_1:%.*]] = add nuw nsw i16 [[TMP1]], 1
; SSE4-NEXT: [[ADD5_1:%.*]] = add nuw nsw i16 [[ADD_1]], [[TMP5]]
+; SSE4-NEXT: [[SHR_1:%.*]] = lshr i16 [[ADD5_1]], 1
; SSE4-NEXT: [[CONV1_6:%.*]] = and i64 [[A_SROA_7_0_EXTRACT_SHIFT]], 255
; SSE4-NEXT: [[CONV4_6:%.*]] = and i64 [[B_SROA_7_0_EXTRACT_SHIFT]], 255
; SSE4-NEXT: [[ADD_6:%.*]] = add nuw nsw i64 [[CONV1_6]], 1
@@ -147,6 +148,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; SSE4-NEXT: [[SHR_8:%.*]] = lshr i64 [[ADD5_8]], 1
; SSE4-NEXT: [[ADD_9:%.*]] = add nuw nsw i16 [[TMP3]], 1
; SSE4-NEXT: [[ADD5_9:%.*]] = add nuw nsw i16 [[ADD_9]], [[TMP7]]
+; SSE4-NEXT: [[SHR_9:%.*]] = lshr i16 [[ADD5_9]], 1
; SSE4-NEXT: [[CONV1_14:%.*]] = and i64 [[A_SROA_16_8_EXTRACT_SHIFT]], 255
; SSE4-NEXT: [[CONV4_14:%.*]] = and i64 [[B_SROA_16_8_EXTRACT_SHIFT]], 255
; SSE4-NEXT: [[ADD_14:%.*]] = add nuw nsw i64 [[CONV1_14]], 1
@@ -156,7 +158,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; SSE4-NEXT: [[TMP8:%.*]] = shl nuw i64 [[ADD5_7]], 55
; SSE4-NEXT: [[RETVAL_SROA_8_0_INSERT_EXT:%.*]] = and i64 [[TMP8]], -72057594037927936
; SSE4-NEXT: [[TMP9:%.*]] = shl nuw nsw i64 [[ADD5_6]], 47
-; SSE4-NEXT: [[RETVAL_SROA_7_0_INSERT_SHIFT:%.*]] = and i64 [[TMP9]], 71776119061217280
+; SSE4-NEXT: [[RETVAL_SROA_7_0_INSERT_SHIFT:%.*]] = and i64 [[TMP9]], 143833713099145216
; SSE4-NEXT: [[TMP10:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_6_0_EXTRACT_SHIFT]], i64 0
; SSE4-NEXT: [[TMP11:%.*]] = insertelement <4 x i64> [[TMP10]], i64 [[A_SROA_5_0_EXTRACT_SHIFT]], i64 1
; SSE4-NEXT: [[TMP12:%.*]] = insertelement <4 x i64> [[TMP11]], i64 [[A_SROA_4_0_EXTRACT_SHIFT]], i64 2
@@ -174,18 +176,17 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; SSE4-NEXT: [[TMP24:%.*]] = bitcast <4 x i32> [[TMP23]] to <16 x i8>
; SSE4-NEXT: [[TMP25:%.*]] = shufflevector <16 x i8> [[TMP24]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
; SSE4-NEXT: [[TMP26:%.*]] = bitcast <8 x i8> [[TMP25]] to i64
-; SSE4-NEXT: [[TMP27:%.*]] = shl nuw i16 [[ADD5_1]], 7
-; SSE4-NEXT: [[TMP28:%.*]] = and i16 [[TMP27]], -256
-; SSE4-NEXT: [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP28]] to i64
-; SSE4-NEXT: [[OP_RDX5:%.*]] = or disjoint i64 [[RETVAL_SROA_8_0_INSERT_EXT]], [[TMP26]]
-; SSE4-NEXT: [[OP_RDX6:%.*]] = or disjoint i64 [[RETVAL_SROA_7_0_INSERT_SHIFT]], [[SHR]]
-; SSE4-NEXT: [[OP_RDX7:%.*]] = or disjoint i64 [[OP_RDX5]], [[OP_RDX6]]
+; SSE4-NEXT: [[RETVAL_SROA_2_0_INSERT_EXT:%.*]] = zext nneg i16 [[SHR_1]] to i64
+; SSE4-NEXT: [[RETVAL_SROA_2_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_2_0_INSERT_EXT]], 8
+; SSE4-NEXT: [[OP_RDX7:%.*]] = or disjoint i64 [[RETVAL_SROA_8_0_INSERT_EXT]], [[TMP26]]
+; SSE4-NEXT: [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED:%.*]] = or disjoint i64 [[RETVAL_SROA_7_0_INSERT_SHIFT]], [[SHR]]
; SSE4-NEXT: [[TMP68:%.*]] = or i64 [[OP_RDX7]], [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED]]
-; SSE4-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP68]], 0
+; SSE4-NEXT: [[OP_RDX8:%.*]] = or i64 [[TMP68]], [[RETVAL_SROA_2_0_INSERT_SHIFT]]
+; SSE4-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[OP_RDX8]], 0
; SSE4-NEXT: [[TMP29:%.*]] = shl nuw i64 [[ADD5_15]], 55
; SSE4-NEXT: [[RETVAL_SROA_17_8_INSERT_EXT:%.*]] = and i64 [[TMP29]], -72057594037927936
; SSE4-NEXT: [[TMP30:%.*]] = shl nuw nsw i64 [[ADD5_14]], 47
-; SSE4-NEXT: [[RETVAL_SROA_16_8_INSERT_SHIFT:%.*]] = and i64 [[TMP30]], 71776119061217280
+; SSE4-NEXT: [[RETVAL_SROA_16_8_INSERT_SHIFT:%.*]] = and i64 [[TMP30]], 143833713099145216
; SSE4-NEXT: [[TMP31:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_15_8_EXTRACT_SHIFT]], i64 0
; SSE4-NEXT: [[TMP32:%.*]] = insertelement <4 x i64> [[TMP31]], i64 [[A_SROA_14_8_EXTRACT_SHIFT]], i64 1
; SSE4-NEXT: [[TMP33:%.*]] = insertelement <4 x i64> [[TMP32]], i64 [[A_SROA_13_8_EXTRACT_SHIFT]], i64 2
@@ -203,14 +204,13 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; SSE4-NEXT: [[TMP45:%.*]] = bitcast <4 x i32> [[TMP44]] to <16 x i8>
; SSE4-NEXT: [[TMP46:%.*]] = shufflevector <16 x i8> [[TMP45]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
; SSE4-NEXT: [[TMP47:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
-; SSE4-NEXT: [[TMP48:%.*]] = shl nuw i16 [[ADD5_9]], 7
-; SSE4-NEXT: [[TMP49:%.*]] = and i16 [[TMP48]], -256
-; SSE4-NEXT: [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP49]] to i64
-; SSE4-NEXT: [[OP_RDX:%.*]] = or disjoint i64 [[RETVAL_SROA_17_8_INSERT_EXT]], [[TMP47]]
-; SSE4-NEXT: [[OP_RDX2:%.*]] = or disjoint i64 [[RETVAL_SROA_16_8_INSERT_SHIFT]], [[SHR_8]]
-; SSE4-NEXT: [[OP_RDX3:%.*]] = or disjoint i64 [[OP_RDX]], [[OP_RDX2]]
+; SSE4-NEXT: [[RETVAL_SROA_11_8_INSERT_EXT:%.*]] = zext nneg i16 [[SHR_9]] to i64
+; SSE4-NEXT: [[RETVAL_SROA_11_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_11_8_INSERT_EXT]], 8
+; SSE4-NEXT: [[OP_RDX3:%.*]] = or disjoint i64 [[RETVAL_SROA_17_8_INSERT_EXT]], [[TMP47]]
+; SSE4-NEXT: [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED:%.*]] = or disjoint i64 [[RETVAL_SROA_16_8_INSERT_SHIFT]], [[SHR_8]]
; SSE4-NEXT: [[TMP69:%.*]] = or i64 [[OP_RDX3]], [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED]]
-; SSE4-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP69]], 1
+; SSE4-NEXT: [[OP_RDX4:%.*]] = or i64 [[TMP69]], [[RETVAL_SROA_11_8_INSERT_SHIFT]]
+; SSE4-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[OP_RDX4]], 1
; SSE4-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
; AVX2-LABEL: @avgr_16_u8(
@@ -223,55 +223,44 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; AVX2-NEXT: [[TMP5:%.*]] = insertelement <2 x i16> poison, i16 [[TMP4]], i64 0
; AVX2-NEXT: [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i16
; AVX2-NEXT: [[TMP7:%.*]] = insertelement <2 x i16> [[TMP5]], i16 [[TMP6]], i64 1
-; AVX2-NEXT: [[CONV1:%.*]] = and i64 [[A_COERCE0]], 255
-; AVX2-NEXT: [[CONV4:%.*]] = and i64 [[B_COERCE0]], 255
-; AVX2-NEXT: [[ADD:%.*]] = add nuw nsw i64 [[CONV1]], 1
-; AVX2-NEXT: [[ADD5:%.*]] = add nuw nsw i64 [[ADD]], [[CONV4]]
-; AVX2-NEXT: [[CONV1_8:%.*]] = and i64 [[A_COERCE1]], 255
-; AVX2-NEXT: [[CONV4_8:%.*]] = and i64 [[B_COERCE1]], 255
-; AVX2-NEXT: [[ADD_8:%.*]] = add nuw nsw i64 [[CONV1_8]], 1
-; AVX2-NEXT: [[ADD5_8:%.*]] = add nuw nsw i64 [[ADD_8]], [[CONV4_8]]
-; AVX2-NEXT: [[TMP8:%.*]] = insertelement <8 x i64> <i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 -1, i64 -1>, i64 [[B_COERCE0]], i64 0
-; AVX2-NEXT: [[TMP9:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 6, i32 7>
+; AVX2-NEXT: [[TMP8:%.*]] = insertelement <8 x i64> <i64 poison, i64 -1, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison>, i64 [[B_COERCE0]], i64 0
+; AVX2-NEXT: [[TMP9:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1>
; AVX2-NEXT: [[TMP10:%.*]] = lshr <8 x i64> [[TMP9]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
; AVX2-NEXT: [[TMP11:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE0]], i64 0
-; AVX2-NEXT: [[TMP12:%.*]] = insertelement <8 x i64> [[TMP11]], i64 [[ADD5]], i64 1
-; AVX2-NEXT: [[TMP13:%.*]] = and <8 x i64> [[TMP10]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0, i64 0>
-; AVX2-NEXT: [[TMP14:%.*]] = insertelement <8 x i64> <i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 -1, i64 -1>, i64 [[B_COERCE1]], i64 0
-; AVX2-NEXT: [[TMP15:%.*]] = shufflevector <8 x i64> [[TMP14]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 6, i32 7>
+; AVX2-NEXT: [[TMP13:%.*]] = and <8 x i64> [[TMP10]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0>
+; AVX2-NEXT: [[TMP14:%.*]] = insertelement <8 x i64> <i64 poison, i64 -1, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison>, i64 [[B_COERCE1]], i64 0
+; AVX2-NEXT: [[TMP15:%.*]] = shufflevector <8 x i64> [[TMP14]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1>
; AVX2-NEXT: [[TMP16:%.*]] = lshr <8 x i64> [[TMP15]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
; AVX2-NEXT: [[TMP17:%.*]] = lshr <2 x i16> [[TMP3]], splat (i16 8)
; AVX2-NEXT: [[TMP18:%.*]] = lshr <2 x i16> [[TMP7]], splat (i16 8)
; AVX2-NEXT: [[TMP19:%.*]] = add nuw nsw <2 x i16> [[TMP17]], splat (i16 1)
; AVX2-NEXT: [[TMP20:%.*]] = add nuw nsw <2 x i16> [[TMP19]], [[TMP18]]
-; AVX2-NEXT: [[TMP21:%.*]] = shl nuw <2 x i16> [[TMP20]], splat (i16 7)
-; AVX2-NEXT: [[TMP22:%.*]] = and <2 x i16> [[TMP21]], splat (i16 -256)
-; AVX2-NEXT: [[TMP23:%.*]] = zext <2 x i16> [[TMP22]] to <2 x i64>
+; AVX2-NEXT: [[TMP21:%.*]] = lshr <2 x i16> [[TMP20]], splat (i16 1)
+; AVX2-NEXT: [[TMP23:%.*]] = zext nneg <2 x i16> [[TMP21]] to <2 x i64>
; AVX2-NEXT: [[TMP24:%.*]] = shufflevector <2 x i64> [[TMP23]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; AVX2-NEXT: [[TMP25:%.*]] = shufflevector <8 x i64> [[TMP12]], <8 x i64> [[TMP24]], <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1, i32 8>
-; AVX2-NEXT: [[TMP26:%.*]] = lshr <8 x i64> [[TMP25]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 1, i64 0>
-; AVX2-NEXT: [[TMP27:%.*]] = and <8 x i64> [[TMP26]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1, i64 -1>
-; AVX2-NEXT: [[TMP28:%.*]] = add nuw nsw <8 x i64> [[TMP27]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
+; AVX2-NEXT: [[TMP26:%.*]] = shufflevector <8 x i64> [[TMP11]], <8 x i64> [[TMP24]], <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 8>
+; AVX2-NEXT: [[TMP27:%.*]] = lshr <8 x i64> [[TMP26]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
+; AVX2-NEXT: [[TMP25:%.*]] = and <8 x i64> [[TMP27]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1>
+; AVX2-NEXT: [[TMP28:%.*]] = add nuw nsw <8 x i64> [[TMP25]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0>
; AVX2-NEXT: [[TMP29:%.*]] = add nuw nsw <8 x i64> [[TMP28]], [[TMP13]]
-; AVX2-NEXT: [[TMP30:%.*]] = trunc <8 x i64> [[TMP29]] to <8 x i32>
-; AVX2-NEXT: [[TMP31:%.*]] = lshr <8 x i32> [[TMP30]], <i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 0, i32 8>
-; AVX2-NEXT: [[TMP41:%.*]] = bitcast <8 x i32> [[TMP31]] to <32 x i8>
-; AVX2-NEXT: [[TMP42:%.*]] = shufflevector <32 x i8> [[TMP41]], <32 x i8> poison, <8 x i32> <i32 24, i32 28, i32 20, i32 16, i32 12, i32 8, i32 4, i32 0>
-; AVX2-NEXT: [[TMP32:%.*]] = bitcast <8 x i8> [[TMP42]] to i64
+; AVX2-NEXT: [[TMP37:%.*]] = shl nuw <8 x i64> [[TMP29]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX2-NEXT: [[TMP44:%.*]] = lshr <8 x i64> [[TMP29]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX2-NEXT: [[TMP30:%.*]] = shufflevector <8 x i64> [[TMP37]], <8 x i64> [[TMP44]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 14, i32 7>
+; AVX2-NEXT: [[TMP31:%.*]] = and <8 x i64> [[TMP30]], <i64 -72057594037927936, i64 143833713099145216, i64 561850441793536, i64 2194728288256, i64 8573157376, i64 33488896, i64 -1, i64 -1>
+; AVX2-NEXT: [[TMP32:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP31]])
; AVX2-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP32]], 0
-; AVX2-NEXT: [[TMP33:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1]], i64 0
-; AVX2-NEXT: [[TMP34:%.*]] = insertelement <8 x i64> [[TMP33]], i64 [[ADD5_8]], i64 1
-; AVX2-NEXT: [[TMP35:%.*]] = shufflevector <8 x i64> [[TMP34]], <8 x i64> [[TMP24]], <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1, i32 9>
-; AVX2-NEXT: [[TMP36:%.*]] = lshr <8 x i64> [[TMP35]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 1, i64 0>
-; AVX2-NEXT: [[TMP37:%.*]] = and <8 x i64> [[TMP36]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1, i64 -1>
-; AVX2-NEXT: [[TMP38:%.*]] = and <8 x i64> [[TMP16]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0, i64 0>
-; AVX2-NEXT: [[TMP39:%.*]] = add nuw nsw <8 x i64> [[TMP37]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
+; AVX2-NEXT: [[TMP33:%.*]] = insertelement <8 x i64> [[TMP24]], i64 [[A_COERCE1]], i64 0
+; AVX2-NEXT: [[TMP34:%.*]] = shufflevector <8 x i64> [[TMP33]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1>
+; AVX2-NEXT: [[TMP35:%.*]] = lshr <8 x i64> [[TMP34]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
+; AVX2-NEXT: [[TMP36:%.*]] = and <8 x i64> [[TMP35]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1>
+; AVX2-NEXT: [[TMP38:%.*]] = and <8 x i64> [[TMP16]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0>
+; AVX2-NEXT: [[TMP39:%.*]] = add nuw nsw <8 x i64> [[TMP36]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0>
; AVX2-NEXT: [[TMP40:%.*]] = add nuw nsw <8 x i64> [[TMP39]], [[TMP38]]
-; AVX2-NEXT: [[TMP47:%.*]] = trunc <8 x i64> [[TMP40]] to <8 x i32>
-; AVX2-NEXT: [[TMP44:%.*]] = lshr <8 x i32> [[TMP47]], <i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 0, i32 8>
-; AVX2-NEXT: [[TMP45:%.*]] = bitcast <8 x i32> [[TMP44]] to <32 x i8>
-; AVX2-NEXT: [[TMP46:%.*]] = shufflevector <32 x i8> [[TMP45]], <32 x i8> poison, <8 x i32> <i32 24, i32 28, i32 20, i32 16, i32 12, i32 8, i32 4, i32 0>
-; AVX2-NEXT: [[TMP43:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
+; AVX2-NEXT: [[TMP45:%.*]] = shl nuw <8 x i64> [[TMP40]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX2-NEXT: [[TMP41:%.*]] = lshr <8 x i64> [[TMP40]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX2-NEXT: [[TMP42:%.*]] = shufflevector <8 x i64> [[TMP45]], <8 x i64> [[TMP41]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 14, i32 7>
+; AVX2-NEXT: [[TMP46:%.*]] = and <8 x i64> [[TMP42]], <i64 -72057594037927936, i64 143833713099145216, i64 561850441793536, i64 2194728288256, i64 8573157376, i64 33488896, i64 -1, i64 -1>
+; AVX2-NEXT: [[TMP43:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP46]])
; AVX2-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP43]], 1
; AVX2-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
@@ -285,55 +274,44 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
; AVX512-NEXT: [[TMP5:%.*]] = insertelement <2 x i16> poison, i16 [[TMP4]], i64 0
; AVX512-NEXT: [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i16
; AVX512-NEXT: [[TMP7:%.*]] = insertelement <2 x i16> [[TMP5]], i16 [[TMP6]], i64 1
-; AVX512-NEXT: [[CONV1:%.*]] = and i64 [[A_COERCE0]], 255
-; AVX512-NEXT: [[CONV4:%.*]] = and i64 [[B_COERCE0]], 255
-; AVX512-NEXT: [[ADD:%.*]] = add nuw nsw i64 [[CONV1]], 1
-; AVX512-NEXT: [[ADD5:%.*]] = add nuw nsw i64 [[ADD]], [[CONV4]]
-; AVX512-NEXT: [[CONV1_8:%.*]] = and i64 [[A_COERCE1]], 255
-; AVX512-NEXT: [[CONV4_8:%.*]] = and i64 [[B_COERCE1]], 255
-; AVX512-NEXT: [[ADD_8:%.*]] = add nuw nsw i64 [[CONV1_8]], 1
-; AVX512-NEXT: [[ADD5_8:%.*]] = add nuw nsw i64 [[ADD_8]], [[CONV4_8]]
-; AVX512-NEXT: [[TMP8:%.*]] = insertelement <8 x i64> <i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 -1, i64 -1>, i64 [[B_COERCE0]], i64 0
-; AVX512-NEXT: [[TMP9:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 6, i32 7>
+; AVX512-NEXT: [[TMP8:%.*]] = insertelement <8 x i64> <i64 poison, i64 -1, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison>, i64 [[B_COERCE0]], i64 0
+; AVX512-NEXT: [[TMP9:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1>
; AVX512-NEXT: [[TMP10:%.*]] = lshr <8 x i64> [[TMP9]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
; AVX512-NEXT: [[TMP11:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE0]], i64 0
-; AVX512-NEXT: [[TMP12:%.*]] = insertelement <8 x i64> [[TMP11]], i64 [[ADD5]], i64 1
-; AVX512-NEXT: [[TMP13:%.*]] = and <8 x i64> [[TMP10]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0, i64 0>
-; AVX512-NEXT: [[TMP14:%.*]] = insertelement <8 x i64> <i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 -1, i64 -1>, i64 [[B_COERCE1]], i64 0
-; AVX512-NEXT: [[TMP15:%.*]] = shufflevector <8 x i64> [[TMP14]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 6, i32 7>
+; AVX512-NEXT: [[TMP13:%.*]] = and <8 x i64> [[TMP10]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0>
+; AVX512-NEXT: [[TMP14:%.*]] = insertelement <8 x i64> <i64 poison, i64 -1, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison>, i64 [[B_COERCE1]], i64 0
+; AVX512-NEXT: [[TMP15:%.*]] = shufflevector <8 x i64> [[TMP14]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1>
; AVX512-NEXT: [[TMP16:%.*]] = lshr <8 x i64> [[TMP15]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
; AVX512-NEXT: [[TMP17:%.*]] = lshr <2 x i16> [[TMP3]], splat (i16 8)
; AVX512-NEXT: [[TMP18:%.*]] = lshr <2 x i16> [[TMP7]], splat (i16 8)
; AVX512-NEXT: [[TMP19:%.*]] = add nuw nsw <2 x i16> [[TMP17]], splat (i16 1)
; AVX512-NEXT: [[TMP20:%.*]] = add nuw nsw <2 x i16> [[TMP19]], [[TMP18]]
-; AVX512-NEXT: [[TMP21:%.*]] = shl nuw <2 x i16> [[TMP20]], splat (i16 7)
-; AVX512-NEXT: [[TMP22:%.*]] = and <2 x i16> [[TMP21]], splat (i16 -256)
+; AVX512-NEXT: [[TMP22:%.*]] = lshr <2 x i16> [[TMP20]], splat (i16 1)
; AVX512-NEXT: [[TMP23:%.*]] = shufflevector <2 x i16> [[TMP22]], <2 x i16> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
; AVX512-NEXT: [[TMP24:%.*]] = zext <8 x i16> [[TMP23]] to <8 x i64>
-; AVX512-NEXT: [[TMP25:%.*]] = shufflevector <8 x i64> [[TMP12]], <8 x i64> [[TMP24]], <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1, i32 8>
-; AVX512-NEXT: [[TMP26:%.*]] = lshr <8 x i64> [[TMP25]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 1, i64 0>
-; AVX512-NEXT: [[TMP27:%.*]] = and <8 x i64> [[TMP26]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1, i64 -1>
-; AVX512-NEXT: [[TMP28:%.*]] = add nuw nsw <8 x i64> [[TMP27]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
+; AVX512-NEXT: [[TMP26:%.*]] = shufflevector <8 x i64> [[TMP11]], <8 x i64> [[TMP24]], <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 8>
+; AVX512-NEXT: [[TMP27:%.*]] = lshr <8 x i64> [[TMP26]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
+; AVX512-NEXT: [[TMP25:%.*]] = and <8 x i64> [[TMP27]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1>
+; AVX512-NEXT: [[TMP28:%.*]] = add nuw nsw <8 x i64> [[TMP25]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0>
; AVX512-NEXT: [[TMP29:%.*]] = add nuw nsw <8 x i64> [[TMP28]], [[TMP13]]
-; AVX512-NEXT: [[TMP30:%.*]] = trunc <8 x i64> [[TMP29]] to <8 x i16>
-; AVX512-NEXT: [[TMP31:%.*]] = lshr <8 x i16> [[TMP30]], <i16 1, i16 1, i16 1, i16 1, i16 1, i16 1, i16 0, i16 8>
-; AVX512-NEXT: [[TMP41:%.*]] = bitcast <8 x i16> [[TMP31]] to <16 x i8>
-; AVX512-NEXT: [[TMP42:%.*]] = shufflevector <16 x i8> [[TMP41]], <16 x i8> poison, <8 x i32> <i32 12, i32 14, i32 10, i32 8, i32 6, i32 4, i32 2, i32 0>
-; AVX512-NEXT: [[TMP32:%.*]] = bitcast <8 x i8> [[TMP42]] to i64
+; AVX512-NEXT: [[TMP37:%.*]] = shl nuw <8 x i64> [[TMP29]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX512-NEXT: [[TMP44:%.*]] = lshr <8 x i64> [[TMP29]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX512-NEXT: [[TMP30:%.*]] = shufflevector <8 x i64> [[TMP37]], <8 x i64> [[TMP44]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 14, i32 7>
+; AVX512-NEXT: [[TMP31:%.*]] = and <8 x i64> [[TMP30]], <i64 -72057594037927936, i64 143833713099145216, i64 561850441793536, i64 2194728288256, i64 8573157376, i64 33488896, i64 -1, i64 -1>
+; AVX512-NEXT: [[TMP32:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP31]])
; AVX512-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP32]], 0
; AVX512-NEXT: [[TMP33:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1]], i64 0
-; AVX512-NEXT: [[TMP34:%.*]] = insertelement <8 x i64> [[TMP33]], i64 [[ADD5_8]], i64 1
-; AVX512-NEXT: [[TMP35:%.*]] = shufflevector <8 x i64> [[TMP34]], <8 x i64> [[TMP24]], <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1, i32 9>
-; AVX512-NEXT: [[TMP36:%.*]] = lshr <8 x i64> [[TMP35]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 1, i64 0>
-; AVX512-NEXT: [[TMP37:%.*]] = and <8 x i64> [[TMP36]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1, i64 -1>
-; AVX512-NEXT: [[TMP38:%.*]] = and <8 x i64> [[TMP16]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0, i64 0>
-; AVX512-NEXT: [[TMP39:%.*]] = add nuw nsw <8 x i64> [[TMP37]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
+; AVX512-NEXT: [[TMP34:%.*]] = shufflevector <8 x i64> [[TMP33]], <8 x i64> [[TMP24]], <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 9>
+; AVX512-NEXT: [[TMP35:%.*]] = lshr <8 x i64> [[TMP34]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
+; AVX512-NEXT: [[TMP36:%.*]] = and <8 x i64> [[TMP35]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1>
+; AVX512-NEXT: [[TMP38:%.*]] = and <8 x i64> [[TMP16]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0>
+; AVX512-NEXT: [[TMP39:%.*]] = add nuw nsw <8 x i64> [[TMP36]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0>
; AVX512-NEXT: [[TMP40:%.*]] = add nuw nsw <8 x i64> [[TMP39]], [[TMP38]]
-; AVX512-NEXT: [[TMP47:%.*]] = trunc <8 x i64> [[TMP40]] to <8 x i16>
-; AVX512-NEXT: [[TMP44:%.*]] = lshr <8 x i16> [[TMP47]], <i16 1, i16 1, i16 1, i16 1, i16 1, i16 1, i16 0, i16 8>
-; AVX512-NEXT: [[TMP45:%.*]] = bitcast <8 x i16> [[TMP44]] to <16 x i8>
-; AVX512-NEXT: [[TMP46:%.*]] = shufflevector <16 x i8> [[TMP45]], <16 x i8> poison, <8 x i32> <i32 12, i32 14, i32 10, i32 8, i32 6, i32 4, i32 2, i32 0>
-; AVX512-NEXT: [[TMP43:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
+; AVX512-NEXT: [[TMP45:%.*]] = shl nuw <8 x i64> [[TMP40]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX512-NEXT: [[TMP41:%.*]] = lshr <8 x i64> [[TMP40]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX512-NEXT: [[TMP42:%.*]] = shufflevector <8 x i64> [[TMP45]], <8 x i64> [[TMP41]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 14, i32 7>
+; AVX512-NEXT: [[TMP46:%.*]] = and <8 x i64> [[TMP42]], <i64 -72057594037927936, i64 143833713099145216, i64 561850441793536, i64 2194728288256, i64 8573157376, i64 33488896, i64 -1, i64 -1>
+; AVX512-NEXT: [[TMP43:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP46]])
; AVX512-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP43]], 1
; AVX512-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
@@ -701,48 +679,185 @@ for.body: ; preds = %for.cond
}
define { i64, i64 } @avgr_8_u16(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0, i64 %b.coerce1) {
-; CHECK-LABEL: @avgr_8_u16(
-; CHECK-NEXT: entry:
-; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
-; CHECK-NEXT: [[TMP2:%.*]] = insertelement <2 x i64> [[TMP1]], i64 [[A_COERCE1:%.*]], i64 1
-; CHECK-NEXT: [[TMP15:%.*]] = trunc <2 x i64> [[TMP2]] to <2 x i32>
-; CHECK-NEXT: [[TMP3:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 32)
-; CHECK-NEXT: [[TMP4:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 48)
-; CHECK-NEXT: [[TMP7:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
-; CHECK-NEXT: [[TMP8:%.*]] = insertelement <2 x i64> [[TMP7]], i64 [[B_COERCE1:%.*]], i64 1
-; CHECK-NEXT: [[TMP18:%.*]] = trunc <2 x i64> [[TMP8]] to <2 x i32>
-; CHECK-NEXT: [[TMP9:%.*]] = lshr <2 x i64> [[TMP8]], splat (i64 32)
-; CHECK-NEXT: [[TMP10:%.*]] = lshr <2 x i64> [[TMP8]], splat (i64 48)
-; CHECK-NEXT: [[TMP12:%.*]] = and <2 x i64> [[TMP2]], splat (i64 65535)
-; CHECK-NEXT: [[TMP13:%.*]] = and <2 x i64> [[TMP8]], splat (i64 65535)
-; CHECK-NEXT: [[TMP16:%.*]] = lshr <2 x i32> [[TMP15]], splat (i32 16)
-; CHECK-NEXT: [[TMP19:%.*]] = lshr <2 x i32> [[TMP18]], splat (i32 16)
-; CHECK-NEXT: [[TMP20:%.*]] = add nuw nsw <2 x i64> [[TMP12]], splat (i64 1)
-; CHECK-NEXT: [[TMP21:%.*]] = add nuw nsw <2 x i64> [[TMP20]], [[TMP13]]
-; CHECK-NEXT: [[TMP22:%.*]] = lshr <2 x i64> [[TMP21]], splat (i64 1)
-; CHECK-NEXT: [[TMP23:%.*]] = add nuw nsw <2 x i32> [[TMP16]], splat (i32 1)
-; CHECK-NEXT: [[TMP24:%.*]] = add nuw nsw <2 x i32> [[TMP23]], [[TMP19]]
-; CHECK-NEXT: [[TMP25:%.*]] = and <2 x i64> [[TMP3]], splat (i64 65535)
-; CHECK-NEXT: [[TMP26:%.*]] = and <2 x i64> [[TMP9]], splat (i64 65535)
-; CHECK-NEXT: [[TMP27:%.*]] = add nuw nsw <2 x i64> [[TMP25]], splat (i64 1)
-; CHECK-NEXT: [[TMP28:%.*]] = add nuw nsw <2 x i64> [[TMP27]], [[TMP26]]
-; CHECK-NEXT: [[TMP29:%.*]] = add nuw nsw <2 x i64> [[TMP4]], splat (i64 1)
-; CHECK-NEXT: [[TMP30:%.*]] = add nuw nsw <2 x i64> [[TMP29]], [[TMP10]]
-; CHECK-NEXT: [[TMP31:%.*]] = shl nuw <2 x i64> [[TMP30]], splat (i64 47)
-; CHECK-NEXT: [[TMP32:%.*]] = and <2 x i64> [[TMP31]], splat (i64 -281474976710656)
-; CHECK-NEXT: [[TMP33:%.*]] = shl nuw nsw <2 x i64> [[TMP28]], splat (i64 31)
-; CHECK-NEXT: [[TMP34:%.*]] = and <2 x i64> [[TMP33]], splat (i64 281470681743360)
-; CHECK-NEXT: [[TMP35:%.*]] = or disjoint <2 x i64> [[TMP32]], [[TMP34]]
-; CHECK-NEXT: [[TMP36:%.*]] = shl nuw <2 x i32> [[TMP24]], splat (i32 15)
-; CHECK-NEXT: [[TMP37:%.*]] = and <2 x i32> [[TMP36]], splat (i32 -65536)
-; CHECK-NEXT: [[TMP38:%.*]] = zext <2 x i32> [[TMP37]] to <2 x i64>
-; CHECK-NEXT: [[TMP39:%.*]] = or disjoint <2 x i64> [[TMP35]], [[TMP38]]
-; CHECK-NEXT: [[TMP40:%.*]] = or disjoint <2 x i64> [[TMP39]], [[TMP22]]
-; CHECK-NEXT: [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP40]], i64 0
-; CHECK-NEXT: [[TMP41:%.*]] = insertvalue { i64, i64 } poison, i64 [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT]], 0
-; CHECK-NEXT: [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP40]], i64 1
-; CHECK-NEXT: [[VEC2STRUCT4:%.*]] = insertvalue { i64, i64 } [[TMP41]], i64 [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT]], 1
-; CHECK-NEXT: ret { i64, i64 } [[VEC2STRUCT4]]
+; SSE2-LABEL: @avgr_8_u16(
+; SSE2-NEXT: entry:
+; SSE2-NEXT: [[TMP0:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
+; SSE2-NEXT: [[TMP1:%.*]] = insertelement <2 x i64> [[TMP0]], i64 [[A_COERCE1:%.*]], i64 1
+; SSE2-NEXT: [[TMP2:%.*]] = trunc <2 x i64> [[TMP1]] to <2 x i32>
+; SSE2-NEXT: [[TMP3:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 32)
+; SSE2-NEXT: [[TMP4:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 48)
+; SSE2-NEXT: [[TMP5:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
+; SSE2-NEXT: [[TMP6:%.*]] = insertelement <2 x i64> [[TMP5]], i64 [[B_COERCE1:%.*]], i64 1
+; SSE2-NEXT: [[TMP7:%.*]] = trunc <2 x i64> [[TMP6]] to <2 x i32>
+; SSE2-NEXT: [[TMP8:%.*]] = lshr <2 x i64> [[TMP6]], splat (i64 32)
+; SSE2-NEXT: [[TMP9:%.*]] = lshr <2 x i64> [[TMP6]], splat (i64 48)
+; SSE2-NEXT: [[TMP10:%.*]] = and <2 x i64> [[TMP1]], splat (i64 65535)
+; SSE2-NEXT: [[TMP11:%.*]] = and <2 x i64> [[TMP6]], splat (i64 65535)
+; SSE2-NEXT: [[TMP12:%.*]] = lshr <2 x i32> [[TMP2]], splat (i32 16)
+; SSE2-NEXT: [[TMP13:%.*]] = lshr <2 x i32> [[TMP7]], splat (i32 16)
+; SSE2-NEXT: [[TMP14:%.*]] = add nuw nsw <2 x i64> [[TMP10]], splat (i64 1)
+; SSE2-NEXT: [[TMP15:%.*]] = add nuw nsw <2 x i64> [[TMP14]], [[TMP11]]
+; SSE2-NEXT: [[TMP16:%.*]] = lshr <2 x i64> [[TMP15]], splat (i64 1)
+; SSE2-NEXT: [[TMP17:%.*]] = add nuw nsw <2 x i32> [[TMP12]], splat (i32 1)
+; SSE2-NEXT: [[TMP18:%.*]] = add nuw nsw <2 x i32> [[TMP17]], [[TMP13]]
+; SSE2-NEXT: [[TMP19:%.*]] = lshr <2 x i32> [[TMP18]], splat (i32 1)
+; SSE2-NEXT: [[TMP20:%.*]] = and <2 x i64> [[TMP3]], splat (i64 65535)
+; SSE2-NEXT: [[TMP21:%.*]] = and <2 x i64> [[TMP8]], splat (i64 65535)
+; SSE2-NEXT: [[TMP22:%.*]] = add nuw nsw <2 x i64> [[TMP20]], splat (i64 1)
+; SSE2-NEXT: [[TMP23:%.*]] = add nuw nsw <2 x i64> [[TMP22]], [[TMP21]]
+; SSE2-NEXT: [[TMP24:%.*]] = add nuw nsw <2 x i64> [[TMP4]], splat (i64 1)
+; SSE2-NEXT: [[TMP25:%.*]] = add nuw nsw <2 x i64> [[TMP24]], [[TMP9]]
+; SSE2-NEXT: [[TMP26:%.*]] = shl nuw <2 x i64> [[TMP25]], splat (i64 47)
+; SSE2-NEXT: [[TMP27:%.*]] = and <2 x i64> [[TMP26]], splat (i64 -281474976710656)
+; SSE2-NEXT: [[TMP28:%.*]] = shl nuw nsw <2 x i64> [[TMP23]], splat (i64 31)
+; SSE2-NEXT: [[TMP29:%.*]] = and <2 x i64> [[TMP28]], splat (i64 562945658454016)
+; SSE2-NEXT: [[TMP30:%.*]] = or <2 x i64> [[TMP27]], [[TMP29]]
+; SSE2-NEXT: [[TMP31:%.*]] = zext nneg <2 x i32> [[TMP19]] to <2 x i64>
+; SSE2-NEXT: [[TMP32:%.*]] = shl nuw nsw <2 x i64> [[TMP31]], splat (i64 16)
+; SSE2-NEXT: [[TMP33:%.*]] = or <2 x i64> [[TMP30]], [[TMP32]]
+; SSE2-NEXT: [[TMP34:%.*]] = or <2 x i64> [[TMP33]], [[TMP16]]
+; SSE2-NEXT: [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP34]], i64 0
+; SSE2-NEXT: [[TMP35:%.*]] = insertvalue { i64, i64 } poison, i64 [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT]], 0
+; SSE2-NEXT: [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP34]], i64 1
+; SSE2-NEXT: [[VEC2STRUCT4:%.*]] = insertvalue { i64, i64 } [[TMP35]], i64 [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT]], 1
+; SSE2-NEXT: ret { i64, i64 } [[VEC2STRUCT4]]
+;
+; SSE4-LABEL: @avgr_8_u16(
+; SSE4-NEXT: entry:
+; SSE4-NEXT: [[TMP0:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
+; SSE4-NEXT: [[TMP1:%.*]] = insertelement <2 x i64> [[TMP0]], i64 [[A_COERCE1:%.*]], i64 1
+; SSE4-NEXT: [[TMP2:%.*]] = trunc <2 x i64> [[TMP1]] to <2 x i32>
+; SSE4-NEXT: [[TMP3:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 32)
+; SSE4-NEXT: [[TMP4:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 48)
+; SSE4-NEXT: [[TMP5:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
+; SSE4-NEXT: [[TMP6:%.*]] = insertelement <2 x i64> [[TMP5]], i64 [[B_COERCE1:%.*]], i64 1
+; SSE4-NEXT: [[TMP7:%.*]] = trunc <2 x i64> [[TMP6]] to <2 x i32>
+; SSE4-NEXT: [[TMP8:%.*]] = lshr <2 x i64> [[TMP6]], splat (i64 32)
+; SSE4-NEXT: [[TMP9:%.*]] = lshr <2 x i64> [[TMP6]], splat (i64 48)
+; SSE4-NEXT: [[TMP10:%.*]] = and <2 x i64> [[TMP1]], splat (i64 65535)
+; SSE4-NEXT: [[TMP11:%.*]] = and <2 x i64> [[TMP6]], splat (i64 65535)
+; SSE4-NEXT: [[TMP12:%.*]] = lshr <2 x i32> [[TMP2]], splat (i32 16)
+; SSE4-NEXT: [[TMP13:%.*]] = lshr <2 x i32> [[TMP7]], splat (i32 16)
+; SSE4-NEXT: [[TMP14:%.*]] = add nuw nsw <2 x i64> [[TMP10]], splat (i64 1)
+; SSE4-NEXT: [[TMP15:%.*]] = add nuw nsw <2 x i64> [[TMP14]], [[TMP11]]
+; SSE4-NEXT: [[TMP16:%.*]] = lshr <2 x i64> [[TMP15]], splat (i64 1)
+; SSE4-NEXT: [[TMP17:%.*]] = add nuw nsw <2 x i32> [[TMP12]], splat (i32 1)
+; SSE4-NEXT: [[TMP18:%.*]] = add nuw nsw <2 x i32> [[TMP17]], [[TMP13]]
+; SSE4-NEXT: [[TMP19:%.*]] = lshr <2 x i32> [[TMP18]], splat (i32 1)
+; SSE4-NEXT: [[TMP20:%.*]] = and <2 x i64> [[TMP3]], splat (i64 65535)
+; SSE4-NEXT: [[TMP21:%.*]] = and <2 x i64> [[TMP8]], splat (i64 65535)
+; SSE4-NEXT: [[TMP22:%.*]] = add nuw nsw <2 x i64> [[TMP20]], splat (i64 1)
+; SSE4-NEXT: [[TMP23:%.*]] = add nuw nsw <2 x i64> [[TMP22]], [[TMP21]]
+; SSE4-NEXT: [[TMP24:%.*]] = add nuw nsw <2 x i64> [[TMP4]], splat (i64 1)
+; SSE4-NEXT: [[TMP25:%.*]] = add nuw nsw <2 x i64> [[TMP24]], [[TMP9]]
+; SSE4-NEXT: [[TMP26:%.*]] = shl nuw <2 x i64> [[TMP25]], splat (i64 47)
+; SSE4-NEXT: [[TMP27:%.*]] = and <2 x i64> [[TMP26]], splat (i64 -281474976710656)
+; SSE4-NEXT: [[TMP28:%.*]] = shl nuw nsw <2 x i64> [[TMP23]], splat (i64 31)
+; SSE4-NEXT: [[TMP29:%.*]] = and <2 x i64> [[TMP28]], splat (i64 562945658454016)
+; SSE4-NEXT: [[TMP30:%.*]] = or <2 x i64> [[TMP27]], [[TMP29]]
+; SSE4-NEXT: [[TMP31:%.*]] = zext nneg <2 x i32> [[TMP19]] to <2 x i64>
+; SSE4-NEXT: [[TMP32:%.*]] = shl nuw nsw <2 x i64> [[TMP31]], splat (i64 16)
+; SSE4-NEXT: [[TMP33:%.*]] = or <2 x i64> [[TMP30]], [[TMP32]]
+; SSE4-NEXT: [[TMP34:%.*]] = or <2 x i64> [[TMP33]], [[TMP16]]
+; SSE4-NEXT: [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP34]], i64 0
+; SSE4-NEXT: [[TMP35:%.*]] = insertvalue { i64, i64 } poison, i64 [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT]], 0
+; SSE4-NEXT: [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP34]], i64 1
+; SSE4-NEXT: [[VEC2STRUCT4:%.*]] = insertvalue { i64, i64 } [[TMP35]], i64 [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT]], 1
+; SSE4-NEXT: ret { i64, i64 } [[VEC2STRUCT4]]
+;
+; AVX2-LABEL: @avgr_8_u16(
+; AVX2-NEXT: entry:
+; AVX2-NEXT: [[TMP0:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
+; AVX2-NEXT: [[TMP1:%.*]] = insertelement <2 x i64> [[TMP0]], i64 [[A_COERCE1:%.*]], i64 1
+; AVX2-NEXT: [[TMP2:%.*]] = trunc <2 x i64> [[TMP1]] to <2 x i32>
+; AVX2-NEXT: [[TMP3:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 32)
+; AVX2-NEXT: [[TMP4:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 48)
+; AVX2-NEXT: [[TMP5:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
+; AVX2-NEXT: [[TMP6:%.*]] = insertelement <2 x i64> [[TMP5]], i64 [[B_COERCE1:%.*]], i64 1
+; AVX2-NEXT: [[TMP7:%.*]] = trunc <2 x i64> [[TMP6]] to <2 x i32>
+; AVX2-NEXT: [[TMP8:%.*]] = lshr <2 x i64> [[TMP6]], splat (i64 32)
+; AVX2-NEXT: [[TMP9:%.*]] = lshr <2 x i64> [[TMP6]], splat (i64 48)
+; AVX2-NEXT: [[TMP10:%.*]] = and <2 x i64> [[TMP1]], splat (i64 65535)
+; AVX2-NEXT: [[TMP11:%.*]] = and <2 x i64> [[TMP6]], splat (i64 65535)
+; AVX2-NEXT: [[TMP12:%.*]] = lshr <2 x i32> [[TMP2]], splat (i32 16)
+; AVX2-NEXT: [[TMP13:%.*]] = lshr <2 x i32> [[TMP7]], splat (i32 16)
+; AVX2-NEXT: [[TMP14:%.*]] = add nuw nsw <2 x i64> [[TMP10]], splat (i64 1)
+; AVX2-NEXT: [[TMP15:%.*]] = add nuw nsw <2 x i64> [[TMP14]], [[TMP11]]
+; AVX2-NEXT: [[TMP16:%.*]] = lshr <2 x i64> [[TMP15]], splat (i64 1)
+; AVX2-NEXT: [[TMP17:%.*]] = add nuw nsw <2 x i32> [[TMP12]], splat (i32 1)
+; AVX2-NEXT: [[TMP18:%.*]] = add nuw nsw <2 x i32> [[TMP17]], [[TMP13]]
+; AVX2-NEXT: [[TMP19:%.*]] = lshr <2 x i32> [[TMP18]], splat (i32 1)
+; AVX2-NEXT: [[TMP20:%.*]] = and <2 x i64> [[TMP3]], splat (i64 65535)
+; AVX2-NEXT: [[TMP21:%.*]] = and <2 x i64> [[TMP8]], splat (i64 65535)
+; AVX2-NEXT: [[TMP22:%.*]] = add nuw nsw <2 x i64> [[TMP20]], splat (i64 1)
+; AVX2-NEXT: [[TMP23:%.*]] = add nuw nsw <2 x i64> [[TMP22]], [[TMP21]]
+; AVX2-NEXT: [[TMP24:%.*]] = add nuw nsw <2 x i64> [[TMP4]], splat (i64 1)
+; AVX2-NEXT: [[TMP25:%.*]] = add nuw nsw <2 x i64> [[TMP24]], [[TMP9]]
+; AVX2-NEXT: [[TMP26:%.*]] = shl nuw <2 x i64> [[TMP25]], splat (i64 47)
+; AVX2-NEXT: [[TMP27:%.*]] = and <2 x i64> [[TMP26]], splat (i64 -281474976710656)
+; AVX2-NEXT: [[TMP28:%.*]] = shl nuw nsw <2 x i64> [[TMP23]], splat (i64 31)
+; AVX2-NEXT: [[TMP29:%.*]] = and <2 x i64> [[TMP28]], splat (i64 562945658454016)
+; AVX2-NEXT: [[TMP30:%.*]] = or <2 x i64> [[TMP27]], [[TMP29]]
+; AVX2-NEXT: [[TMP31:%.*]] = zext nneg <2 x i32> [[TMP19]] to <2 x i64>
+; AVX2-NEXT: [[TMP32:%.*]] = shl nuw nsw <2 x i64> [[TMP31]], splat (i64 16)
+; AVX2-NEXT: [[TMP33:%.*]] = or <2 x i64> [[TMP30]], [[TMP32]]
+; AVX2-NEXT: [[TMP34:%.*]] = or <2 x i64> [[TMP33]], [[TMP16]]
+; AVX2-NEXT: [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP34]], i64 0
+; AVX2-NEXT: [[TMP35:%.*]] = insertvalue { i64, i64 } poison, i64 [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT]], 0
+; AVX2-NEXT: [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP34]], i64 1
+; AVX2-NEXT: [[VEC2STRUCT4:%.*]] = insertvalue { i64, i64 } [[TMP35]], i64 [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT]], 1
+; AVX2-NEXT: ret { i64, i64 } [[VEC2STRUCT4]]
+;
+; AVX512-LABEL: @avgr_8_u16(
+; AVX512-NEXT: entry:
+; AVX512-NEXT: [[TMP0:%.*]] = trunc i64 [[A_COERCE0:%.*]] to i32
+; AVX512-NEXT: [[TMP1:%.*]] = insertelement <2 x i32> poison, i32 [[TMP0]], i64 0
+; AVX512-NEXT: [[TMP2:%.*]] = trunc i64 [[A_COERCE1:%.*]] to i32
+; AVX512-NEXT: [[TMP3:%.*]] = insertelement <2 x i32> [[TMP1]], i32 [[TMP2]], i64 1
+; AVX512-NEXT: [[TMP4:%.*]] = trunc i64 [[B_COERCE0:%.*]] to i32
+; AVX512-NEXT: [[TMP5:%.*]] = insertelement <2 x i32> poison, i32 [[TMP4]], i64 0
+; AVX512-NEXT: [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i32
+; AVX512-NEXT: [[TMP7:%.*]] = insertelement <2 x i32> [[TMP5]], i32 [[TMP6]], i64 1
+; AVX512-NEXT: [[TMP8:%.*]] = insertelement <4 x i64> <i64 poison, i64 -1, i64 poison, i64 poison>, i64 [[B_COERCE0]], i64 0
+; AVX512-NEXT: [[TMP9:%.*]] = shufflevector <4 x i64> [[TMP8]], <4 x i64> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
+; AVX512-NEXT: [[TMP10:%.*]] = lshr <4 x i64> [[TMP9]], <i64 48, i64 32, i64 0, i64 0>
+; AVX512-NEXT: [[TMP11:%.*]] = insertelement <4 x i64> poison, i64 [[A_COERCE0]], i64 0
+; AVX512-NEXT: [[TMP12:%.*]] = and <4 x i64> [[TMP10]], <i64 -1, i64 65535, i64 65535, i64 0>
+; AVX512-NEXT: [[TMP13:%.*]] = insertelement <4 x i64> <i64 poison, i64 -1, i64 poison, i64 poison>, i64 [[B_COERCE1]], i64 0
+; AVX512-NEXT: [[TMP14:%.*]] = shufflevector <4 x i64> [[TMP13]], <4 x i64> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
+; AVX512-NEXT: [[TMP15:%.*]] = lshr <4 x i64> [[TMP14]], <i64 48, i64 32, i64 0, i64 0>
+; AVX512-NEXT: [[TMP16:%.*]] = lshr <2 x i32> [[TMP3]], splat (i32 16)
+; AVX512-NEXT: [[TMP17:%.*]] = lshr <2 x i32> [[TMP7]], splat (i32 16)
+; AVX512-NEXT: [[TMP18:%.*]] = add nuw nsw <2 x i32> [[TMP16]], splat (i32 1)
+; AVX512-NEXT: [[TMP19:%.*]] = add nuw nsw <2 x i32> [[TMP18]], [[TMP17]]
+; AVX512-NEXT: [[TMP20:%.*]] = lshr <2 x i32> [[TMP19]], splat (i32 1)
+; AVX512-NEXT: [[TMP21:%.*]] = shufflevector <2 x i32> [[TMP20]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; AVX512-NEXT: [[TMP22:%.*]] = zext <4 x i32> [[TMP21]] to <4 x i64>
+; AVX512-NEXT: [[TMP23:%.*]] = shufflevector <4 x i64> [[TMP11]], <4 x i64> [[TMP22]], <4 x i32> <i32 0, i32 0, i32 0, i32 4>
+; AVX512-NEXT: [[TMP24:%.*]] = lshr <4 x i64> [[TMP23]], <i64 48, i64 32, i64 0, i64 0>
+; AVX512-NEXT: [[TMP25:%.*]] = and <4 x i64> [[TMP24]], <i64 -1, i64 65535, i64 65535, i64 -1>
+; AVX512-NEXT: [[TMP26:%.*]] = add nuw nsw <4 x i64> [[TMP25]], <i64 1, i64 1, i64 1, i64 0>
+; AVX512-NEXT: [[TMP27:%.*]] = add nuw nsw <4 x i64> [[TMP26]], [[TMP12]]
+; AVX512-NEXT: [[TMP28:%.*]] = shl nuw <4 x i64> [[TMP27]], <i64 47, i64 31, i64 1, i64 16>
+; AVX512-NEXT: [[TMP29:%.*]] = lshr <4 x i64> [[TMP27]], <i64 47, i64 31, i64 1, i64 16>
+; AVX512-NEXT: [[TMP30:%.*]] = shufflevector <4 x i64> [[TMP28]], <4 x i64> [[TMP29]], <4 x i32> <i32 0, i32 1, i32 6, i32 3>
+; AVX512-NEXT: [[TMP31:%.*]] = and <4 x i64> [[TMP30]], <i64 -281474976710656, i64 562945658454016, i64 -1, i64 -1>
+; AVX512-NEXT: [[TMP32:%.*]] = tail call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP31]])
+; AVX512-NEXT: [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP32]], 0
+; AVX512-NEXT: [[TMP33:%.*]] = insertelement <4 x i64> poison, i64 [[A_COERCE1]], i64 0
+; AVX512-NEXT: [[TMP34:%.*]] = shufflevector <4 x i64> [[TMP33]], <4 x i64> [[TMP22]], <4 x i32> <i32 0, i32 0, i32 0, i32 5>
+; AVX512-NEXT: [[TMP35:%.*]] = lshr <4 x i64> [[TMP34]], <i64 48, i64 32, i64 0, i64 0>
+; AVX512-NEXT: [[TMP36:%.*]] = and <4 x i64> [[TMP35]], <i64 -1, i64 65535, i64 65535, i64 -1>
+; AVX512-NEXT: [[TMP37:%.*]] = and <4 x i64> [[TMP15]], <i64 -1, i64 65535, i64 65535, i64 0>
+; AVX512-NEXT: [[TMP38:%.*]] = add nuw nsw <4 x i64> [[TMP36]], <i64 1, i64 1, i64 1, i64 0>
+; AVX512-NEXT: [[TMP39:%.*]] = add nuw nsw <4 x i64> [[TMP38]], [[TMP37]]
+; AVX512-NEXT: [[TMP40:%.*]] = shl nuw <4 x i64> [[TMP39]], <i64 47, i64 31, i64 1, i64 16>
+; AVX512-NEXT: [[TMP41:%.*]] = lshr <4 x i64> [[TMP39]], <i64 47, i64 31, i64 1, i64 16>
+; AVX512-NEXT: [[TMP42:%.*]] = shufflevector <4 x i64> [[TMP40]], <4 x i64> [[TMP41]], <4 x i32> <i32 0, i32 1, i32 6, i32 3>
+; AVX512-NEXT: [[TMP43:%.*]] = and <4 x i64> [[TMP42]], <i64 -281474976710656, i64 562945658454016, i64 -1, i64 -1>
+; AVX512-NEXT: [[TMP44:%.*]] = tail call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP43]])
+; AVX512-NEXT: [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP44]], 1
+; AVX512-NEXT: ret { i64, i64 } [[DOTFCA_1_INSERT]]
;
entry:
%retval = alloca %"struct.std::array8", align 2
@@ -1033,3 +1148,5 @@ for.body: ; preds = %for.cond
%inc = add nuw nsw i64 %i.0, 1
br label %for.cond
}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK: {{.*}}
More information about the llvm-commits
mailing list