[llvm] [InstCombine] Optimize `zext(trunc nuw x)` (PR #225206)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 16:53:33 PDT 2026


https://github.com/peterbell10 updated https://github.com/llvm/llvm-project/pull/225206

>From 0172007d4587facadd271cb459e0fda99da0a2e9 Mon Sep 17 00:00:00 2001
From: Peter Bell <peterbell10 at openai.com>
Date: Mon, 21 Sep 2026 22:38:21 +0100
Subject: [PATCH 1/2] [InstCombine] Fold zext(trunc nuw x)

---
 .../InstCombine/InstCombineCasts.cpp          |  26 +++--
 .../InstCombine/sext-of-trunc-nsw.ll          |  40 +++++++
 llvm/test/Transforms/InstCombine/trunc.ll     |   6 +-
 llvm/test/Transforms/InstCombine/zext.ll      | 104 ++++++++++++++++++
 4 files changed, 165 insertions(+), 11 deletions(-)

diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
index 97defc5e3ddd3..274eed449dc59 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCasts.cpp
@@ -239,6 +239,23 @@ Instruction *InstCombinerImpl::commonCastTransforms(CastInst &CI) {
         replaceAllDbgUsesWith(*CSrc, *Res, CI, DT);
       return Res;
     }
+
+    // A no-wrap trunc followed by the corresponding extension preserves the
+    // value, so cast directly to the final type.
+    bool IsZExt = isa<ZExtInst>(CI);
+    bool IsSExt = isa<SExtInst>(CI);
+    if (auto *Trunc = dyn_cast<TruncInst>(CSrc);
+        Trunc && ((IsZExt && Trunc->hasNoUnsignedWrap()) ||
+                  (IsSExt && Trunc->hasNoSignedWrap()))) {
+      auto *Res = CastInst::CreateIntegerCast(Trunc->getOperand(0), Ty, IsSExt);
+      if (auto *ResTrunc = dyn_cast<TruncInst>(Res)) {
+        ResTrunc->setHasNoUnsignedWrap(Trunc->hasNoUnsignedWrap());
+        ResTrunc->setHasNoSignedWrap(Trunc->hasNoSignedWrap());
+      } else if (auto *ResZExt = dyn_cast<ZExtInst>(Res)) {
+        ResZExt->setNonNeg(true);
+      }
+      return Res;
+    }
   }
 
   if (auto *Sel = dyn_cast<SelectInst>(Src)) {
@@ -1967,13 +1984,8 @@ Instruction *InstCombinerImpl::visitSExt(SExtInst &Sext) {
     // If the input has more sign bits than bits truncated, then convert
     // directly to final type.
     unsigned XBitSize = X->getType()->getScalarSizeInBits();
-    bool HasNSW = cast<TruncInst>(Src)->hasNoSignedWrap();
-    if (HasNSW || (ComputeNumSignBits(X, &Sext) > XBitSize - SrcBitSize)) {
-      auto *Res = CastInst::CreateIntegerCast(X, DestTy, /* isSigned */ true);
-      if (auto *ResTrunc = dyn_cast<TruncInst>(Res); ResTrunc && HasNSW)
-        ResTrunc->setHasNoSignedWrap(true);
-      return Res;
-    }
+    if (ComputeNumSignBits(X, &Sext) > XBitSize - SrcBitSize)
+      return CastInst::CreateIntegerCast(X, DestTy, /* isSigned */ true);
 
     // If input is a trunc from the destination type, then convert into shifts.
     if (Src->hasOneUse() && X->getType() == DestTy) {
diff --git a/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll b/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll
index 4128a15d8d7ce..d6fb3a10fccd6 100644
--- a/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll
+++ b/llvm/test/Transforms/InstCombine/sext-of-trunc-nsw.ll
@@ -233,3 +233,43 @@ define i32 @same_source_not_matching_signbits_extra_use(i32 %x) {
   %c = sext i8 %b to i32
   ret i32 %c
 }
+
+define i32 @sext_trunc_widen(i8 %x) {
+; CHECK-LABEL: @sext_trunc_widen(
+; CHECK-NEXT:    [[RESULT:%.*]] = sext i8 [[X:%.*]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc nsw i8 %x to i1
+  %result = sext i1 %narrow to i32
+  ret i32 %result
+}
+
+define i8 @sext_trunc_same(i8 %x) {
+; CHECK-LABEL: @sext_trunc_same(
+; CHECK-NEXT:    ret i8 [[X:%.*]]
+;
+  %narrow = trunc nsw i8 %x to i1
+  %result = sext i1 %narrow to i8
+  ret i8 %result
+}
+
+define i32 @sext_trunc_narrow(i64 %x) {
+; CHECK-LABEL: @sext_trunc_narrow(
+; CHECK-NEXT:    [[RESULT:%.*]] = trunc nsw i64 [[X:%.*]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc nsw i64 %x to i1
+  %result = sext i1 %narrow to i32
+  ret i32 %result
+}
+
+define i32 @sext_trunc_nuw(i8 %x) {
+; CHECK-LABEL: @sext_trunc_nuw(
+; CHECK-NEXT:    [[NARROW:%.*]] = trunc nuw i8 [[X:%.*]] to i1
+; CHECK-NEXT:    [[RESULT:%.*]] = sext i1 [[NARROW]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc nuw i8 %x to i1
+  %result = sext i1 %narrow to i32
+  ret i32 %result
+}
diff --git a/llvm/test/Transforms/InstCombine/trunc.ll b/llvm/test/Transforms/InstCombine/trunc.ll
index 91f637e89069f..6f99bf444ccb4 100644
--- a/llvm/test/Transforms/InstCombine/trunc.ll
+++ b/llvm/test/Transforms/InstCombine/trunc.ll
@@ -1406,8 +1406,7 @@ define i32 @neg_zext_i32_trunc_nsw_i8(i16 %x, i32 %y) {
 
 define i16 @zext_i16_trunc_nuw_nsw_i8(i32 %x) {
 ; CHECK-LABEL: @zext_i16_trunc_nuw_nsw_i8(
-; CHECK-NEXT:    [[C:%.*]] = trunc nuw nsw i32 [[X:%.*]] to i16
-; CHECK-NEXT:    [[E:%.*]] = and i16 [[C]], 255
+; CHECK-NEXT:    [[E:%.*]] = trunc nuw nsw i32 [[X:%.*]] to i16
 ; CHECK-NEXT:    ret i16 [[E]]
 ;
   %c = trunc nuw nsw i32 %x to i8
@@ -1428,8 +1427,7 @@ define i16 @zext_i16_trunc_nsw_i8(i32 %x) {
 
 define i16 @zext_i16_trunc_nuw_i8(i32 %x) {
 ; CHECK-LABEL: @zext_i16_trunc_nuw_i8(
-; CHECK-NEXT:    [[C:%.*]] = trunc nuw i32 [[X:%.*]] to i16
-; CHECK-NEXT:    [[E:%.*]] = and i16 [[C]], 255
+; CHECK-NEXT:    [[E:%.*]] = trunc nuw i32 [[X:%.*]] to i16
 ; CHECK-NEXT:    ret i16 [[E]]
 ;
   %c = trunc nuw i32 %x to i8
diff --git a/llvm/test/Transforms/InstCombine/zext.ll b/llvm/test/Transforms/InstCombine/zext.ll
index 3fb1e77ae2335..e98710d6b43d1 100644
--- a/llvm/test/Transforms/InstCombine/zext.ll
+++ b/llvm/test/Transforms/InstCombine/zext.ll
@@ -1080,3 +1080,107 @@ define <2 x i8> @zext_or_trunc_nuw_vec(<2 x i8> %x, <2 x i4> %y) {
   %zext = zext <2 x i4> %or to <2 x i8>
   ret <2 x i8> %zext
 }
+
+define i32 @zext_trunc_widen(i8 %x) {
+; CHECK-LABEL: @zext_trunc_widen(
+; CHECK-NEXT:    [[RESULT:%.*]] = zext nneg i8 [[X:%.*]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc nuw i8 %x to i1
+  %result = zext i1 %narrow to i32
+  ret i32 %result
+}
+
+define i64 @zext_trunc_native(i8 %x) {
+; CHECK-LABEL: @zext_trunc_native(
+; CHECK-NEXT:    [[RESULT:%.*]] = zext nneg i8 [[X:%.*]] to i64
+; CHECK-NEXT:    ret i64 [[RESULT]]
+;
+  %narrow = trunc nuw i8 %x to i1
+  %result = zext i1 %narrow to i64
+  ret i64 %result
+}
+
+define i8 @zext_trunc_same(i8 %x) {
+; CHECK-LABEL: @zext_trunc_same(
+; CHECK-NEXT:    ret i8 [[X:%.*]]
+;
+  %narrow = trunc nuw i8 %x to i1
+  %result = zext i1 %narrow to i8
+  ret i8 %result
+}
+
+define i32 @zext_trunc_narrow(i64 %x) {
+; CHECK-LABEL: @zext_trunc_narrow(
+; CHECK-NEXT:    [[RESULT:%.*]] = trunc nuw i64 [[X:%.*]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc nuw i64 %x to i1
+  %result = zext i1 %narrow to i32
+  ret i32 %result
+}
+
+define i32 @zext_trunc_no_flag(i8 %x) {
+; CHECK-LABEL: @zext_trunc_no_flag(
+; CHECK-NEXT:    [[NARROW_MASK:%.*]] = and i8 [[X:%.*]], 1
+; CHECK-NEXT:    [[RESULT:%.*]] = zext nneg i8 [[NARROW_MASK]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc i8 %x to i1
+  %result = zext i1 %narrow to i32
+  ret i32 %result
+}
+
+define i32 @zext_trunc_nsw(i8 %x) {
+; CHECK-LABEL: @zext_trunc_nsw(
+; CHECK-NEXT:    [[NARROW_MASK:%.*]] = and i8 [[X:%.*]], 1
+; CHECK-NEXT:    [[RESULT:%.*]] = zext nneg i8 [[NARROW_MASK]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc nsw i8 %x to i1
+  %result = zext i1 %narrow to i32
+  ret i32 %result
+}
+
+define <2 x i32> @zext_trunc_nuw_vec(<2 x i8> %x) {
+; CHECK-LABEL: @zext_trunc_nuw_vec(
+; CHECK-NEXT:    [[RESULT:%.*]] = zext nneg <2 x i8> [[X:%.*]] to <2 x i32>
+; CHECK-NEXT:    ret <2 x i32> [[RESULT]]
+;
+  %narrow = trunc nuw <2 x i8> %x to <2 x i1>
+  %result = zext <2 x i1> %narrow to <2 x i32>
+  ret <2 x i32> %result
+}
+
+define <vscale x 2 x i32> @zext_trunc_nuw_scalable(<vscale x 2 x i8> %x) {
+; CHECK-LABEL: @zext_trunc_nuw_scalable(
+; CHECK-NEXT:    [[RESULT:%.*]] = zext nneg <vscale x 2 x i8> [[X:%.*]] to <vscale x 2 x i32>
+; CHECK-NEXT:    ret <vscale x 2 x i32> [[RESULT]]
+;
+  %narrow = trunc nuw <vscale x 2 x i8> %x to <vscale x 2 x i1>
+  %result = zext <vscale x 2 x i1> %narrow to <vscale x 2 x i32>
+  ret <vscale x 2 x i32> %result
+}
+
+define <2 x i32> @zext_trunc_nuw_vec_narrow(<2 x i64> %x) {
+; CHECK-LABEL: @zext_trunc_nuw_vec_narrow(
+; CHECK-NEXT:    [[RESULT:%.*]] = trunc nuw <2 x i64> [[X:%.*]] to <2 x i32>
+; CHECK-NEXT:    ret <2 x i32> [[RESULT]]
+;
+  %narrow = trunc nuw <2 x i64> %x to <2 x i1>
+  %result = zext <2 x i1> %narrow to <2 x i32>
+  ret <2 x i32> %result
+}
+
+define i32 @zext_trunc_nuw_multi_use(i8 %x) {
+; CHECK-LABEL: @zext_trunc_nuw_multi_use(
+; CHECK-NEXT:    [[NARROW:%.*]] = trunc nuw i8 [[X:%.*]] to i1
+; CHECK-NEXT:    call void @use1(i1 [[NARROW]])
+; CHECK-NEXT:    [[RESULT:%.*]] = zext nneg i8 [[X]] to i32
+; CHECK-NEXT:    ret i32 [[RESULT]]
+;
+  %narrow = trunc nuw i8 %x to i1
+  call void @use1(i1 %narrow)
+  %result = zext i1 %narrow to i32
+  ret i32 %result
+}

>From c00534f56e70f66572d8de1ffc79eefcaa731ad6 Mon Sep 17 00:00:00 2001
From: Peter Bell <peterbell10 at openai.com>
Date: Tue, 22 Sep 2026 00:52:35 +0100
Subject: [PATCH 2/2] Fix x86 lit test

---
 llvm/test/Transforms/PhaseOrdering/X86/avg.ll | 409 +++++++++++-------
 1 file changed, 263 insertions(+), 146 deletions(-)

diff --git a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
index 8f8899d285ad3..7abb52200084a 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/avg.ll
@@ -44,6 +44,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE2-NEXT:    [[TMP26:%.*]] = lshr <2 x i64> [[TMP25]], splat (i64 1)
 ; SSE2-NEXT:    [[TMP27:%.*]] = add nuw nsw <2 x i16> [[TMP20]], splat (i16 1)
 ; SSE2-NEXT:    [[TMP28:%.*]] = add nuw nsw <2 x i16> [[TMP27]], [[TMP23]]
+; SSE2-NEXT:    [[TMP50:%.*]] = lshr <2 x i16> [[TMP28]], splat (i16 1)
 ; SSE2-NEXT:    [[TMP29:%.*]] = and <2 x i64> [[TMP3]], splat (i64 255)
 ; SSE2-NEXT:    [[TMP30:%.*]] = and <2 x i64> [[TMP11]], splat (i64 255)
 ; SSE2-NEXT:    [[TMP31:%.*]] = add nuw nsw <2 x i64> [[TMP29]], splat (i64 1)
@@ -68,25 +69,24 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE2-NEXT:    [[TMP75:%.*]] = shl nuw <2 x i64> [[TMP74]], splat (i64 55)
 ; SSE2-NEXT:    [[TMP76:%.*]] = and <2 x i64> [[TMP75]], splat (i64 -72057594037927936)
 ; SSE2-NEXT:    [[TMP77:%.*]] = shl nuw nsw <2 x i64> [[TMP72]], splat (i64 47)
-; SSE2-NEXT:    [[TMP78:%.*]] = and <2 x i64> [[TMP77]], splat (i64 71776119061217280)
-; SSE2-NEXT:    [[TMP52:%.*]] = or disjoint <2 x i64> [[TMP76]], [[TMP78]]
+; SSE2-NEXT:    [[TMP53:%.*]] = and <2 x i64> [[TMP77]], splat (i64 143833713099145216)
+; SSE2-NEXT:    [[TMP55:%.*]] = or <2 x i64> [[TMP76]], [[TMP53]]
 ; SSE2-NEXT:    [[TMP49:%.*]] = shl nuw nsw <2 x i64> [[TMP44]], splat (i64 39)
-; SSE2-NEXT:    [[TMP50:%.*]] = and <2 x i64> [[TMP49]], splat (i64 280375465082880)
-; SSE2-NEXT:    [[TMP53:%.*]] = or disjoint <2 x i64> [[TMP52]], [[TMP50]]
+; SSE2-NEXT:    [[TMP56:%.*]] = and <2 x i64> [[TMP49]], splat (i64 561850441793536)
+; SSE2-NEXT:    [[TMP58:%.*]] = or <2 x i64> [[TMP55]], [[TMP56]]
 ; SSE2-NEXT:    [[TMP54:%.*]] = shl nuw nsw <2 x i64> [[TMP40]], splat (i64 31)
-; SSE2-NEXT:    [[TMP55:%.*]] = and <2 x i64> [[TMP54]], splat (i64 1095216660480)
-; SSE2-NEXT:    [[TMP56:%.*]] = or disjoint <2 x i64> [[TMP53]], [[TMP55]]
+; SSE2-NEXT:    [[TMP59:%.*]] = and <2 x i64> [[TMP54]], splat (i64 2194728288256)
+; SSE2-NEXT:    [[TMP61:%.*]] = or <2 x i64> [[TMP58]], [[TMP59]]
 ; SSE2-NEXT:    [[TMP57:%.*]] = shl nuw nsw <2 x i64> [[TMP36]], splat (i64 23)
-; SSE2-NEXT:    [[TMP58:%.*]] = and <2 x i64> [[TMP57]], splat (i64 4278190080)
-; SSE2-NEXT:    [[TMP59:%.*]] = or disjoint <2 x i64> [[TMP56]], [[TMP58]]
+; SSE2-NEXT:    [[TMP62:%.*]] = and <2 x i64> [[TMP57]], splat (i64 8573157376)
+; SSE2-NEXT:    [[TMP63:%.*]] = or <2 x i64> [[TMP61]], [[TMP62]]
 ; SSE2-NEXT:    [[TMP60:%.*]] = shl nuw nsw <2 x i64> [[TMP32]], splat (i64 15)
-; SSE2-NEXT:    [[TMP61:%.*]] = and <2 x i64> [[TMP60]], splat (i64 16711680)
-; SSE2-NEXT:    [[TMP62:%.*]] = shl nuw <2 x i16> [[TMP28]], splat (i16 7)
-; SSE2-NEXT:    [[TMP63:%.*]] = or disjoint <2 x i64> [[TMP59]], [[TMP61]]
-; SSE2-NEXT:    [[TMP64:%.*]] = and <2 x i16> [[TMP62]], splat (i16 -256)
-; SSE2-NEXT:    [[TMP65:%.*]] = zext <2 x i16> [[TMP64]] to <2 x i64>
+; SSE2-NEXT:    [[TMP65:%.*]] = and <2 x i64> [[TMP60]], splat (i64 33488896)
+; SSE2-NEXT:    [[TMP78:%.*]] = zext nneg <2 x i16> [[TMP50]] to <2 x i64>
+; SSE2-NEXT:    [[TMP79:%.*]] = shl nuw nsw <2 x i64> [[TMP78]], splat (i64 8)
 ; SSE2-NEXT:    [[TMP66:%.*]] = or <2 x i64> [[TMP63]], [[TMP65]]
-; SSE2-NEXT:    [[TMP67:%.*]] = or <2 x i64> [[TMP66]], [[TMP26]]
+; SSE2-NEXT:    [[TMP80:%.*]] = or <2 x i64> [[TMP66]], [[TMP79]]
+; SSE2-NEXT:    [[TMP67:%.*]] = or <2 x i64> [[TMP80]], [[TMP26]]
 ; SSE2-NEXT:    [[TMP68:%.*]] = extractelement <2 x i64> [[TMP67]], i64 0
 ; SSE2-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP68]], 0
 ; SSE2-NEXT:    [[TMP69:%.*]] = extractelement <2 x i64> [[TMP67]], i64 1
@@ -134,6 +134,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE4-NEXT:    [[SHR:%.*]] = lshr i64 [[ADD5]], 1
 ; SSE4-NEXT:    [[ADD_1:%.*]] = add nuw nsw i16 [[TMP1]], 1
 ; SSE4-NEXT:    [[ADD5_1:%.*]] = add nuw nsw i16 [[ADD_1]], [[TMP5]]
+; SSE4-NEXT:    [[SHR_1:%.*]] = lshr i16 [[ADD5_1]], 1
 ; SSE4-NEXT:    [[CONV1_6:%.*]] = and i64 [[A_SROA_7_0_EXTRACT_SHIFT]], 255
 ; SSE4-NEXT:    [[CONV4_6:%.*]] = and i64 [[B_SROA_7_0_EXTRACT_SHIFT]], 255
 ; SSE4-NEXT:    [[ADD_6:%.*]] = add nuw nsw i64 [[CONV1_6]], 1
@@ -147,6 +148,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE4-NEXT:    [[SHR_8:%.*]] = lshr i64 [[ADD5_8]], 1
 ; SSE4-NEXT:    [[ADD_9:%.*]] = add nuw nsw i16 [[TMP3]], 1
 ; SSE4-NEXT:    [[ADD5_9:%.*]] = add nuw nsw i16 [[ADD_9]], [[TMP7]]
+; SSE4-NEXT:    [[SHR_9:%.*]] = lshr i16 [[ADD5_9]], 1
 ; SSE4-NEXT:    [[CONV1_14:%.*]] = and i64 [[A_SROA_16_8_EXTRACT_SHIFT]], 255
 ; SSE4-NEXT:    [[CONV4_14:%.*]] = and i64 [[B_SROA_16_8_EXTRACT_SHIFT]], 255
 ; SSE4-NEXT:    [[ADD_14:%.*]] = add nuw nsw i64 [[CONV1_14]], 1
@@ -156,7 +158,7 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE4-NEXT:    [[TMP8:%.*]] = shl nuw i64 [[ADD5_7]], 55
 ; SSE4-NEXT:    [[RETVAL_SROA_8_0_INSERT_EXT:%.*]] = and i64 [[TMP8]], -72057594037927936
 ; SSE4-NEXT:    [[TMP9:%.*]] = shl nuw nsw i64 [[ADD5_6]], 47
-; SSE4-NEXT:    [[RETVAL_SROA_7_0_INSERT_SHIFT:%.*]] = and i64 [[TMP9]], 71776119061217280
+; SSE4-NEXT:    [[RETVAL_SROA_7_0_INSERT_SHIFT:%.*]] = and i64 [[TMP9]], 143833713099145216
 ; SSE4-NEXT:    [[TMP10:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_6_0_EXTRACT_SHIFT]], i64 0
 ; SSE4-NEXT:    [[TMP11:%.*]] = insertelement <4 x i64> [[TMP10]], i64 [[A_SROA_5_0_EXTRACT_SHIFT]], i64 1
 ; SSE4-NEXT:    [[TMP12:%.*]] = insertelement <4 x i64> [[TMP11]], i64 [[A_SROA_4_0_EXTRACT_SHIFT]], i64 2
@@ -174,18 +176,17 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE4-NEXT:    [[TMP24:%.*]] = bitcast <4 x i32> [[TMP23]] to <16 x i8>
 ; SSE4-NEXT:    [[TMP25:%.*]] = shufflevector <16 x i8> [[TMP24]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
 ; SSE4-NEXT:    [[TMP26:%.*]] = bitcast <8 x i8> [[TMP25]] to i64
-; SSE4-NEXT:    [[TMP27:%.*]] = shl nuw i16 [[ADD5_1]], 7
-; SSE4-NEXT:    [[TMP28:%.*]] = and i16 [[TMP27]], -256
-; SSE4-NEXT:    [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP28]] to i64
-; SSE4-NEXT:    [[OP_RDX5:%.*]] = or disjoint i64 [[RETVAL_SROA_8_0_INSERT_EXT]], [[TMP26]]
-; SSE4-NEXT:    [[OP_RDX6:%.*]] = or disjoint i64 [[RETVAL_SROA_7_0_INSERT_SHIFT]], [[SHR]]
-; SSE4-NEXT:    [[OP_RDX7:%.*]] = or disjoint i64 [[OP_RDX5]], [[OP_RDX6]]
+; SSE4-NEXT:    [[RETVAL_SROA_2_0_INSERT_EXT:%.*]] = zext nneg i16 [[SHR_1]] to i64
+; SSE4-NEXT:    [[RETVAL_SROA_2_0_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_2_0_INSERT_EXT]], 8
+; SSE4-NEXT:    [[OP_RDX7:%.*]] = or disjoint i64 [[RETVAL_SROA_8_0_INSERT_EXT]], [[TMP26]]
+; SSE4-NEXT:    [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED:%.*]] = or disjoint i64 [[RETVAL_SROA_7_0_INSERT_SHIFT]], [[SHR]]
 ; SSE4-NEXT:    [[TMP68:%.*]] = or i64 [[OP_RDX7]], [[RETVAL_SROA_2_0_INSERT_SHIFT_MASKED]]
-; SSE4-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP68]], 0
+; SSE4-NEXT:    [[OP_RDX8:%.*]] = or i64 [[TMP68]], [[RETVAL_SROA_2_0_INSERT_SHIFT]]
+; SSE4-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[OP_RDX8]], 0
 ; SSE4-NEXT:    [[TMP29:%.*]] = shl nuw i64 [[ADD5_15]], 55
 ; SSE4-NEXT:    [[RETVAL_SROA_17_8_INSERT_EXT:%.*]] = and i64 [[TMP29]], -72057594037927936
 ; SSE4-NEXT:    [[TMP30:%.*]] = shl nuw nsw i64 [[ADD5_14]], 47
-; SSE4-NEXT:    [[RETVAL_SROA_16_8_INSERT_SHIFT:%.*]] = and i64 [[TMP30]], 71776119061217280
+; SSE4-NEXT:    [[RETVAL_SROA_16_8_INSERT_SHIFT:%.*]] = and i64 [[TMP30]], 143833713099145216
 ; SSE4-NEXT:    [[TMP31:%.*]] = insertelement <4 x i64> poison, i64 [[A_SROA_15_8_EXTRACT_SHIFT]], i64 0
 ; SSE4-NEXT:    [[TMP32:%.*]] = insertelement <4 x i64> [[TMP31]], i64 [[A_SROA_14_8_EXTRACT_SHIFT]], i64 1
 ; SSE4-NEXT:    [[TMP33:%.*]] = insertelement <4 x i64> [[TMP32]], i64 [[A_SROA_13_8_EXTRACT_SHIFT]], i64 2
@@ -203,14 +204,13 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; SSE4-NEXT:    [[TMP45:%.*]] = bitcast <4 x i32> [[TMP44]] to <16 x i8>
 ; SSE4-NEXT:    [[TMP46:%.*]] = shufflevector <16 x i8> [[TMP45]], <16 x i8> <i8 0, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison, i8 poison>, <8 x i32> <i32 16, i32 16, i32 12, i32 8, i32 4, i32 0, i32 16, i32 16>
 ; SSE4-NEXT:    [[TMP47:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
-; SSE4-NEXT:    [[TMP48:%.*]] = shl nuw i16 [[ADD5_9]], 7
-; SSE4-NEXT:    [[TMP49:%.*]] = and i16 [[TMP48]], -256
-; SSE4-NEXT:    [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED:%.*]] = zext i16 [[TMP49]] to i64
-; SSE4-NEXT:    [[OP_RDX:%.*]] = or disjoint i64 [[RETVAL_SROA_17_8_INSERT_EXT]], [[TMP47]]
-; SSE4-NEXT:    [[OP_RDX2:%.*]] = or disjoint i64 [[RETVAL_SROA_16_8_INSERT_SHIFT]], [[SHR_8]]
-; SSE4-NEXT:    [[OP_RDX3:%.*]] = or disjoint i64 [[OP_RDX]], [[OP_RDX2]]
+; SSE4-NEXT:    [[RETVAL_SROA_11_8_INSERT_EXT:%.*]] = zext nneg i16 [[SHR_9]] to i64
+; SSE4-NEXT:    [[RETVAL_SROA_11_8_INSERT_SHIFT:%.*]] = shl nuw nsw i64 [[RETVAL_SROA_11_8_INSERT_EXT]], 8
+; SSE4-NEXT:    [[OP_RDX3:%.*]] = or disjoint i64 [[RETVAL_SROA_17_8_INSERT_EXT]], [[TMP47]]
+; SSE4-NEXT:    [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED:%.*]] = or disjoint i64 [[RETVAL_SROA_16_8_INSERT_SHIFT]], [[SHR_8]]
 ; SSE4-NEXT:    [[TMP69:%.*]] = or i64 [[OP_RDX3]], [[RETVAL_SROA_11_8_INSERT_SHIFT_MASKED]]
-; SSE4-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP69]], 1
+; SSE4-NEXT:    [[OP_RDX4:%.*]] = or i64 [[TMP69]], [[RETVAL_SROA_11_8_INSERT_SHIFT]]
+; SSE4-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[OP_RDX4]], 1
 ; SSE4-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
 ; AVX2-LABEL: @avgr_16_u8(
@@ -223,55 +223,44 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; AVX2-NEXT:    [[TMP5:%.*]] = insertelement <2 x i16> poison, i16 [[TMP4]], i64 0
 ; AVX2-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i16
 ; AVX2-NEXT:    [[TMP7:%.*]] = insertelement <2 x i16> [[TMP5]], i16 [[TMP6]], i64 1
-; AVX2-NEXT:    [[CONV1:%.*]] = and i64 [[A_COERCE0]], 255
-; AVX2-NEXT:    [[CONV4:%.*]] = and i64 [[B_COERCE0]], 255
-; AVX2-NEXT:    [[ADD:%.*]] = add nuw nsw i64 [[CONV1]], 1
-; AVX2-NEXT:    [[ADD5:%.*]] = add nuw nsw i64 [[ADD]], [[CONV4]]
-; AVX2-NEXT:    [[CONV1_8:%.*]] = and i64 [[A_COERCE1]], 255
-; AVX2-NEXT:    [[CONV4_8:%.*]] = and i64 [[B_COERCE1]], 255
-; AVX2-NEXT:    [[ADD_8:%.*]] = add nuw nsw i64 [[CONV1_8]], 1
-; AVX2-NEXT:    [[ADD5_8:%.*]] = add nuw nsw i64 [[ADD_8]], [[CONV4_8]]
-; AVX2-NEXT:    [[TMP8:%.*]] = insertelement <8 x i64> <i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 -1, i64 -1>, i64 [[B_COERCE0]], i64 0
-; AVX2-NEXT:    [[TMP9:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 6, i32 7>
+; AVX2-NEXT:    [[TMP8:%.*]] = insertelement <8 x i64> <i64 poison, i64 -1, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison>, i64 [[B_COERCE0]], i64 0
+; AVX2-NEXT:    [[TMP9:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1>
 ; AVX2-NEXT:    [[TMP10:%.*]] = lshr <8 x i64> [[TMP9]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
 ; AVX2-NEXT:    [[TMP11:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE0]], i64 0
-; AVX2-NEXT:    [[TMP12:%.*]] = insertelement <8 x i64> [[TMP11]], i64 [[ADD5]], i64 1
-; AVX2-NEXT:    [[TMP13:%.*]] = and <8 x i64> [[TMP10]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0, i64 0>
-; AVX2-NEXT:    [[TMP14:%.*]] = insertelement <8 x i64> <i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 -1, i64 -1>, i64 [[B_COERCE1]], i64 0
-; AVX2-NEXT:    [[TMP15:%.*]] = shufflevector <8 x i64> [[TMP14]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 6, i32 7>
+; AVX2-NEXT:    [[TMP13:%.*]] = and <8 x i64> [[TMP10]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0>
+; AVX2-NEXT:    [[TMP14:%.*]] = insertelement <8 x i64> <i64 poison, i64 -1, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison>, i64 [[B_COERCE1]], i64 0
+; AVX2-NEXT:    [[TMP15:%.*]] = shufflevector <8 x i64> [[TMP14]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1>
 ; AVX2-NEXT:    [[TMP16:%.*]] = lshr <8 x i64> [[TMP15]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
 ; AVX2-NEXT:    [[TMP17:%.*]] = lshr <2 x i16> [[TMP3]], splat (i16 8)
 ; AVX2-NEXT:    [[TMP18:%.*]] = lshr <2 x i16> [[TMP7]], splat (i16 8)
 ; AVX2-NEXT:    [[TMP19:%.*]] = add nuw nsw <2 x i16> [[TMP17]], splat (i16 1)
 ; AVX2-NEXT:    [[TMP20:%.*]] = add nuw nsw <2 x i16> [[TMP19]], [[TMP18]]
-; AVX2-NEXT:    [[TMP21:%.*]] = shl nuw <2 x i16> [[TMP20]], splat (i16 7)
-; AVX2-NEXT:    [[TMP22:%.*]] = and <2 x i16> [[TMP21]], splat (i16 -256)
-; AVX2-NEXT:    [[TMP23:%.*]] = zext <2 x i16> [[TMP22]] to <2 x i64>
+; AVX2-NEXT:    [[TMP21:%.*]] = lshr <2 x i16> [[TMP20]], splat (i16 1)
+; AVX2-NEXT:    [[TMP23:%.*]] = zext nneg <2 x i16> [[TMP21]] to <2 x i64>
 ; AVX2-NEXT:    [[TMP24:%.*]] = shufflevector <2 x i64> [[TMP23]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; AVX2-NEXT:    [[TMP25:%.*]] = shufflevector <8 x i64> [[TMP12]], <8 x i64> [[TMP24]], <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1, i32 8>
-; AVX2-NEXT:    [[TMP26:%.*]] = lshr <8 x i64> [[TMP25]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 1, i64 0>
-; AVX2-NEXT:    [[TMP27:%.*]] = and <8 x i64> [[TMP26]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1, i64 -1>
-; AVX2-NEXT:    [[TMP28:%.*]] = add nuw nsw <8 x i64> [[TMP27]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
+; AVX2-NEXT:    [[TMP26:%.*]] = shufflevector <8 x i64> [[TMP11]], <8 x i64> [[TMP24]], <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 8>
+; AVX2-NEXT:    [[TMP27:%.*]] = lshr <8 x i64> [[TMP26]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
+; AVX2-NEXT:    [[TMP25:%.*]] = and <8 x i64> [[TMP27]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1>
+; AVX2-NEXT:    [[TMP28:%.*]] = add nuw nsw <8 x i64> [[TMP25]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0>
 ; AVX2-NEXT:    [[TMP29:%.*]] = add nuw nsw <8 x i64> [[TMP28]], [[TMP13]]
-; AVX2-NEXT:    [[TMP30:%.*]] = trunc <8 x i64> [[TMP29]] to <8 x i32>
-; AVX2-NEXT:    [[TMP31:%.*]] = lshr <8 x i32> [[TMP30]], <i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 0, i32 8>
-; AVX2-NEXT:    [[TMP41:%.*]] = bitcast <8 x i32> [[TMP31]] to <32 x i8>
-; AVX2-NEXT:    [[TMP42:%.*]] = shufflevector <32 x i8> [[TMP41]], <32 x i8> poison, <8 x i32> <i32 24, i32 28, i32 20, i32 16, i32 12, i32 8, i32 4, i32 0>
-; AVX2-NEXT:    [[TMP32:%.*]] = bitcast <8 x i8> [[TMP42]] to i64
+; AVX2-NEXT:    [[TMP37:%.*]] = shl nuw <8 x i64> [[TMP29]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX2-NEXT:    [[TMP44:%.*]] = lshr <8 x i64> [[TMP29]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX2-NEXT:    [[TMP30:%.*]] = shufflevector <8 x i64> [[TMP37]], <8 x i64> [[TMP44]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 14, i32 7>
+; AVX2-NEXT:    [[TMP31:%.*]] = and <8 x i64> [[TMP30]], <i64 -72057594037927936, i64 143833713099145216, i64 561850441793536, i64 2194728288256, i64 8573157376, i64 33488896, i64 -1, i64 -1>
+; AVX2-NEXT:    [[TMP32:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP31]])
 ; AVX2-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP32]], 0
-; AVX2-NEXT:    [[TMP33:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1]], i64 0
-; AVX2-NEXT:    [[TMP34:%.*]] = insertelement <8 x i64> [[TMP33]], i64 [[ADD5_8]], i64 1
-; AVX2-NEXT:    [[TMP35:%.*]] = shufflevector <8 x i64> [[TMP34]], <8 x i64> [[TMP24]], <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1, i32 9>
-; AVX2-NEXT:    [[TMP36:%.*]] = lshr <8 x i64> [[TMP35]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 1, i64 0>
-; AVX2-NEXT:    [[TMP37:%.*]] = and <8 x i64> [[TMP36]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1, i64 -1>
-; AVX2-NEXT:    [[TMP38:%.*]] = and <8 x i64> [[TMP16]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0, i64 0>
-; AVX2-NEXT:    [[TMP39:%.*]] = add nuw nsw <8 x i64> [[TMP37]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
+; AVX2-NEXT:    [[TMP33:%.*]] = insertelement <8 x i64> [[TMP24]], i64 [[A_COERCE1]], i64 0
+; AVX2-NEXT:    [[TMP34:%.*]] = shufflevector <8 x i64> [[TMP33]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1>
+; AVX2-NEXT:    [[TMP35:%.*]] = lshr <8 x i64> [[TMP34]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
+; AVX2-NEXT:    [[TMP36:%.*]] = and <8 x i64> [[TMP35]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1>
+; AVX2-NEXT:    [[TMP38:%.*]] = and <8 x i64> [[TMP16]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0>
+; AVX2-NEXT:    [[TMP39:%.*]] = add nuw nsw <8 x i64> [[TMP36]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0>
 ; AVX2-NEXT:    [[TMP40:%.*]] = add nuw nsw <8 x i64> [[TMP39]], [[TMP38]]
-; AVX2-NEXT:    [[TMP47:%.*]] = trunc <8 x i64> [[TMP40]] to <8 x i32>
-; AVX2-NEXT:    [[TMP44:%.*]] = lshr <8 x i32> [[TMP47]], <i32 1, i32 1, i32 1, i32 1, i32 1, i32 1, i32 0, i32 8>
-; AVX2-NEXT:    [[TMP45:%.*]] = bitcast <8 x i32> [[TMP44]] to <32 x i8>
-; AVX2-NEXT:    [[TMP46:%.*]] = shufflevector <32 x i8> [[TMP45]], <32 x i8> poison, <8 x i32> <i32 24, i32 28, i32 20, i32 16, i32 12, i32 8, i32 4, i32 0>
-; AVX2-NEXT:    [[TMP43:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
+; AVX2-NEXT:    [[TMP45:%.*]] = shl nuw <8 x i64> [[TMP40]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX2-NEXT:    [[TMP41:%.*]] = lshr <8 x i64> [[TMP40]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX2-NEXT:    [[TMP42:%.*]] = shufflevector <8 x i64> [[TMP45]], <8 x i64> [[TMP41]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 14, i32 7>
+; AVX2-NEXT:    [[TMP46:%.*]] = and <8 x i64> [[TMP42]], <i64 -72057594037927936, i64 143833713099145216, i64 561850441793536, i64 2194728288256, i64 8573157376, i64 33488896, i64 -1, i64 -1>
+; AVX2-NEXT:    [[TMP43:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP46]])
 ; AVX2-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP43]], 1
 ; AVX2-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
@@ -285,55 +274,44 @@ define { i64, i64 } @avgr_16_u8(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0,
 ; AVX512-NEXT:    [[TMP5:%.*]] = insertelement <2 x i16> poison, i16 [[TMP4]], i64 0
 ; AVX512-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i16
 ; AVX512-NEXT:    [[TMP7:%.*]] = insertelement <2 x i16> [[TMP5]], i16 [[TMP6]], i64 1
-; AVX512-NEXT:    [[CONV1:%.*]] = and i64 [[A_COERCE0]], 255
-; AVX512-NEXT:    [[CONV4:%.*]] = and i64 [[B_COERCE0]], 255
-; AVX512-NEXT:    [[ADD:%.*]] = add nuw nsw i64 [[CONV1]], 1
-; AVX512-NEXT:    [[ADD5:%.*]] = add nuw nsw i64 [[ADD]], [[CONV4]]
-; AVX512-NEXT:    [[CONV1_8:%.*]] = and i64 [[A_COERCE1]], 255
-; AVX512-NEXT:    [[CONV4_8:%.*]] = and i64 [[B_COERCE1]], 255
-; AVX512-NEXT:    [[ADD_8:%.*]] = add nuw nsw i64 [[CONV1_8]], 1
-; AVX512-NEXT:    [[ADD5_8:%.*]] = add nuw nsw i64 [[ADD_8]], [[CONV4_8]]
-; AVX512-NEXT:    [[TMP8:%.*]] = insertelement <8 x i64> <i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 -1, i64 -1>, i64 [[B_COERCE0]], i64 0
-; AVX512-NEXT:    [[TMP9:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 6, i32 7>
+; AVX512-NEXT:    [[TMP8:%.*]] = insertelement <8 x i64> <i64 poison, i64 -1, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison>, i64 [[B_COERCE0]], i64 0
+; AVX512-NEXT:    [[TMP9:%.*]] = shufflevector <8 x i64> [[TMP8]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1>
 ; AVX512-NEXT:    [[TMP10:%.*]] = lshr <8 x i64> [[TMP9]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
 ; AVX512-NEXT:    [[TMP11:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE0]], i64 0
-; AVX512-NEXT:    [[TMP12:%.*]] = insertelement <8 x i64> [[TMP11]], i64 [[ADD5]], i64 1
-; AVX512-NEXT:    [[TMP13:%.*]] = and <8 x i64> [[TMP10]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0, i64 0>
-; AVX512-NEXT:    [[TMP14:%.*]] = insertelement <8 x i64> <i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 -1, i64 -1>, i64 [[B_COERCE1]], i64 0
-; AVX512-NEXT:    [[TMP15:%.*]] = shufflevector <8 x i64> [[TMP14]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 6, i32 7>
+; AVX512-NEXT:    [[TMP13:%.*]] = and <8 x i64> [[TMP10]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0>
+; AVX512-NEXT:    [[TMP14:%.*]] = insertelement <8 x i64> <i64 poison, i64 -1, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison>, i64 [[B_COERCE1]], i64 0
+; AVX512-NEXT:    [[TMP15:%.*]] = shufflevector <8 x i64> [[TMP14]], <8 x i64> poison, <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1>
 ; AVX512-NEXT:    [[TMP16:%.*]] = lshr <8 x i64> [[TMP15]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
 ; AVX512-NEXT:    [[TMP17:%.*]] = lshr <2 x i16> [[TMP3]], splat (i16 8)
 ; AVX512-NEXT:    [[TMP18:%.*]] = lshr <2 x i16> [[TMP7]], splat (i16 8)
 ; AVX512-NEXT:    [[TMP19:%.*]] = add nuw nsw <2 x i16> [[TMP17]], splat (i16 1)
 ; AVX512-NEXT:    [[TMP20:%.*]] = add nuw nsw <2 x i16> [[TMP19]], [[TMP18]]
-; AVX512-NEXT:    [[TMP21:%.*]] = shl nuw <2 x i16> [[TMP20]], splat (i16 7)
-; AVX512-NEXT:    [[TMP22:%.*]] = and <2 x i16> [[TMP21]], splat (i16 -256)
+; AVX512-NEXT:    [[TMP22:%.*]] = lshr <2 x i16> [[TMP20]], splat (i16 1)
 ; AVX512-NEXT:    [[TMP23:%.*]] = shufflevector <2 x i16> [[TMP22]], <2 x i16> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
 ; AVX512-NEXT:    [[TMP24:%.*]] = zext <8 x i16> [[TMP23]] to <8 x i64>
-; AVX512-NEXT:    [[TMP25:%.*]] = shufflevector <8 x i64> [[TMP12]], <8 x i64> [[TMP24]], <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1, i32 8>
-; AVX512-NEXT:    [[TMP26:%.*]] = lshr <8 x i64> [[TMP25]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 1, i64 0>
-; AVX512-NEXT:    [[TMP27:%.*]] = and <8 x i64> [[TMP26]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1, i64 -1>
-; AVX512-NEXT:    [[TMP28:%.*]] = add nuw nsw <8 x i64> [[TMP27]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
+; AVX512-NEXT:    [[TMP26:%.*]] = shufflevector <8 x i64> [[TMP11]], <8 x i64> [[TMP24]], <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 8>
+; AVX512-NEXT:    [[TMP27:%.*]] = lshr <8 x i64> [[TMP26]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
+; AVX512-NEXT:    [[TMP25:%.*]] = and <8 x i64> [[TMP27]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1>
+; AVX512-NEXT:    [[TMP28:%.*]] = add nuw nsw <8 x i64> [[TMP25]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0>
 ; AVX512-NEXT:    [[TMP29:%.*]] = add nuw nsw <8 x i64> [[TMP28]], [[TMP13]]
-; AVX512-NEXT:    [[TMP30:%.*]] = trunc <8 x i64> [[TMP29]] to <8 x i16>
-; AVX512-NEXT:    [[TMP31:%.*]] = lshr <8 x i16> [[TMP30]], <i16 1, i16 1, i16 1, i16 1, i16 1, i16 1, i16 0, i16 8>
-; AVX512-NEXT:    [[TMP41:%.*]] = bitcast <8 x i16> [[TMP31]] to <16 x i8>
-; AVX512-NEXT:    [[TMP42:%.*]] = shufflevector <16 x i8> [[TMP41]], <16 x i8> poison, <8 x i32> <i32 12, i32 14, i32 10, i32 8, i32 6, i32 4, i32 2, i32 0>
-; AVX512-NEXT:    [[TMP32:%.*]] = bitcast <8 x i8> [[TMP42]] to i64
+; AVX512-NEXT:    [[TMP37:%.*]] = shl nuw <8 x i64> [[TMP29]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX512-NEXT:    [[TMP44:%.*]] = lshr <8 x i64> [[TMP29]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX512-NEXT:    [[TMP30:%.*]] = shufflevector <8 x i64> [[TMP37]], <8 x i64> [[TMP44]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 14, i32 7>
+; AVX512-NEXT:    [[TMP31:%.*]] = and <8 x i64> [[TMP30]], <i64 -72057594037927936, i64 143833713099145216, i64 561850441793536, i64 2194728288256, i64 8573157376, i64 33488896, i64 -1, i64 -1>
+; AVX512-NEXT:    [[TMP32:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP31]])
 ; AVX512-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP32]], 0
 ; AVX512-NEXT:    [[TMP33:%.*]] = insertelement <8 x i64> poison, i64 [[A_COERCE1]], i64 0
-; AVX512-NEXT:    [[TMP34:%.*]] = insertelement <8 x i64> [[TMP33]], i64 [[ADD5_8]], i64 1
-; AVX512-NEXT:    [[TMP35:%.*]] = shufflevector <8 x i64> [[TMP34]], <8 x i64> [[TMP24]], <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 1, i32 9>
-; AVX512-NEXT:    [[TMP36:%.*]] = lshr <8 x i64> [[TMP35]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 1, i64 0>
-; AVX512-NEXT:    [[TMP37:%.*]] = and <8 x i64> [[TMP36]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1, i64 -1>
-; AVX512-NEXT:    [[TMP38:%.*]] = and <8 x i64> [[TMP16]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0, i64 0>
-; AVX512-NEXT:    [[TMP39:%.*]] = add nuw nsw <8 x i64> [[TMP37]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0, i64 0>
+; AVX512-NEXT:    [[TMP34:%.*]] = shufflevector <8 x i64> [[TMP33]], <8 x i64> [[TMP24]], <8 x i32> <i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 9>
+; AVX512-NEXT:    [[TMP35:%.*]] = lshr <8 x i64> [[TMP34]], <i64 56, i64 48, i64 40, i64 32, i64 24, i64 16, i64 0, i64 0>
+; AVX512-NEXT:    [[TMP36:%.*]] = and <8 x i64> [[TMP35]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 -1>
+; AVX512-NEXT:    [[TMP38:%.*]] = and <8 x i64> [[TMP16]], <i64 -1, i64 255, i64 255, i64 255, i64 255, i64 255, i64 255, i64 0>
+; AVX512-NEXT:    [[TMP39:%.*]] = add nuw nsw <8 x i64> [[TMP36]], <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 0>
 ; AVX512-NEXT:    [[TMP40:%.*]] = add nuw nsw <8 x i64> [[TMP39]], [[TMP38]]
-; AVX512-NEXT:    [[TMP47:%.*]] = trunc <8 x i64> [[TMP40]] to <8 x i16>
-; AVX512-NEXT:    [[TMP44:%.*]] = lshr <8 x i16> [[TMP47]], <i16 1, i16 1, i16 1, i16 1, i16 1, i16 1, i16 0, i16 8>
-; AVX512-NEXT:    [[TMP45:%.*]] = bitcast <8 x i16> [[TMP44]] to <16 x i8>
-; AVX512-NEXT:    [[TMP46:%.*]] = shufflevector <16 x i8> [[TMP45]], <16 x i8> poison, <8 x i32> <i32 12, i32 14, i32 10, i32 8, i32 6, i32 4, i32 2, i32 0>
-; AVX512-NEXT:    [[TMP43:%.*]] = bitcast <8 x i8> [[TMP46]] to i64
+; AVX512-NEXT:    [[TMP45:%.*]] = shl nuw <8 x i64> [[TMP40]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX512-NEXT:    [[TMP41:%.*]] = lshr <8 x i64> [[TMP40]], <i64 55, i64 47, i64 39, i64 31, i64 23, i64 15, i64 1, i64 8>
+; AVX512-NEXT:    [[TMP42:%.*]] = shufflevector <8 x i64> [[TMP45]], <8 x i64> [[TMP41]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 14, i32 7>
+; AVX512-NEXT:    [[TMP46:%.*]] = and <8 x i64> [[TMP42]], <i64 -72057594037927936, i64 143833713099145216, i64 561850441793536, i64 2194728288256, i64 8573157376, i64 33488896, i64 -1, i64 -1>
+; AVX512-NEXT:    [[TMP43:%.*]] = tail call i64 @llvm.vector.reduce.or.v8i64(<8 x i64> [[TMP46]])
 ; AVX512-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP43]], 1
 ; AVX512-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
@@ -701,48 +679,185 @@ for.body:                                         ; preds = %for.cond
 }
 
 define { i64, i64 } @avgr_8_u16(i64 %a.coerce0, i64 %a.coerce1, i64 %b.coerce0, i64 %b.coerce1) {
-; CHECK-LABEL: @avgr_8_u16(
-; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
-; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <2 x i64> [[TMP1]], i64 [[A_COERCE1:%.*]], i64 1
-; CHECK-NEXT:    [[TMP15:%.*]] = trunc <2 x i64> [[TMP2]] to <2 x i32>
-; CHECK-NEXT:    [[TMP3:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 32)
-; CHECK-NEXT:    [[TMP4:%.*]] = lshr <2 x i64> [[TMP2]], splat (i64 48)
-; CHECK-NEXT:    [[TMP7:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
-; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <2 x i64> [[TMP7]], i64 [[B_COERCE1:%.*]], i64 1
-; CHECK-NEXT:    [[TMP18:%.*]] = trunc <2 x i64> [[TMP8]] to <2 x i32>
-; CHECK-NEXT:    [[TMP9:%.*]] = lshr <2 x i64> [[TMP8]], splat (i64 32)
-; CHECK-NEXT:    [[TMP10:%.*]] = lshr <2 x i64> [[TMP8]], splat (i64 48)
-; CHECK-NEXT:    [[TMP12:%.*]] = and <2 x i64> [[TMP2]], splat (i64 65535)
-; CHECK-NEXT:    [[TMP13:%.*]] = and <2 x i64> [[TMP8]], splat (i64 65535)
-; CHECK-NEXT:    [[TMP16:%.*]] = lshr <2 x i32> [[TMP15]], splat (i32 16)
-; CHECK-NEXT:    [[TMP19:%.*]] = lshr <2 x i32> [[TMP18]], splat (i32 16)
-; CHECK-NEXT:    [[TMP20:%.*]] = add nuw nsw <2 x i64> [[TMP12]], splat (i64 1)
-; CHECK-NEXT:    [[TMP21:%.*]] = add nuw nsw <2 x i64> [[TMP20]], [[TMP13]]
-; CHECK-NEXT:    [[TMP22:%.*]] = lshr <2 x i64> [[TMP21]], splat (i64 1)
-; CHECK-NEXT:    [[TMP23:%.*]] = add nuw nsw <2 x i32> [[TMP16]], splat (i32 1)
-; CHECK-NEXT:    [[TMP24:%.*]] = add nuw nsw <2 x i32> [[TMP23]], [[TMP19]]
-; CHECK-NEXT:    [[TMP25:%.*]] = and <2 x i64> [[TMP3]], splat (i64 65535)
-; CHECK-NEXT:    [[TMP26:%.*]] = and <2 x i64> [[TMP9]], splat (i64 65535)
-; CHECK-NEXT:    [[TMP27:%.*]] = add nuw nsw <2 x i64> [[TMP25]], splat (i64 1)
-; CHECK-NEXT:    [[TMP28:%.*]] = add nuw nsw <2 x i64> [[TMP27]], [[TMP26]]
-; CHECK-NEXT:    [[TMP29:%.*]] = add nuw nsw <2 x i64> [[TMP4]], splat (i64 1)
-; CHECK-NEXT:    [[TMP30:%.*]] = add nuw nsw <2 x i64> [[TMP29]], [[TMP10]]
-; CHECK-NEXT:    [[TMP31:%.*]] = shl nuw <2 x i64> [[TMP30]], splat (i64 47)
-; CHECK-NEXT:    [[TMP32:%.*]] = and <2 x i64> [[TMP31]], splat (i64 -281474976710656)
-; CHECK-NEXT:    [[TMP33:%.*]] = shl nuw nsw <2 x i64> [[TMP28]], splat (i64 31)
-; CHECK-NEXT:    [[TMP34:%.*]] = and <2 x i64> [[TMP33]], splat (i64 281470681743360)
-; CHECK-NEXT:    [[TMP35:%.*]] = or disjoint <2 x i64> [[TMP32]], [[TMP34]]
-; CHECK-NEXT:    [[TMP36:%.*]] = shl nuw <2 x i32> [[TMP24]], splat (i32 15)
-; CHECK-NEXT:    [[TMP37:%.*]] = and <2 x i32> [[TMP36]], splat (i32 -65536)
-; CHECK-NEXT:    [[TMP38:%.*]] = zext <2 x i32> [[TMP37]] to <2 x i64>
-; CHECK-NEXT:    [[TMP39:%.*]] = or disjoint <2 x i64> [[TMP35]], [[TMP38]]
-; CHECK-NEXT:    [[TMP40:%.*]] = or disjoint <2 x i64> [[TMP39]], [[TMP22]]
-; CHECK-NEXT:    [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP40]], i64 0
-; CHECK-NEXT:    [[TMP41:%.*]] = insertvalue { i64, i64 } poison, i64 [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT]], 0
-; CHECK-NEXT:    [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP40]], i64 1
-; CHECK-NEXT:    [[VEC2STRUCT4:%.*]] = insertvalue { i64, i64 } [[TMP41]], i64 [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT]], 1
-; CHECK-NEXT:    ret { i64, i64 } [[VEC2STRUCT4]]
+; SSE2-LABEL: @avgr_8_u16(
+; SSE2-NEXT:  entry:
+; SSE2-NEXT:    [[TMP0:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
+; SSE2-NEXT:    [[TMP1:%.*]] = insertelement <2 x i64> [[TMP0]], i64 [[A_COERCE1:%.*]], i64 1
+; SSE2-NEXT:    [[TMP2:%.*]] = trunc <2 x i64> [[TMP1]] to <2 x i32>
+; SSE2-NEXT:    [[TMP3:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 32)
+; SSE2-NEXT:    [[TMP4:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 48)
+; SSE2-NEXT:    [[TMP5:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
+; SSE2-NEXT:    [[TMP6:%.*]] = insertelement <2 x i64> [[TMP5]], i64 [[B_COERCE1:%.*]], i64 1
+; SSE2-NEXT:    [[TMP7:%.*]] = trunc <2 x i64> [[TMP6]] to <2 x i32>
+; SSE2-NEXT:    [[TMP8:%.*]] = lshr <2 x i64> [[TMP6]], splat (i64 32)
+; SSE2-NEXT:    [[TMP9:%.*]] = lshr <2 x i64> [[TMP6]], splat (i64 48)
+; SSE2-NEXT:    [[TMP10:%.*]] = and <2 x i64> [[TMP1]], splat (i64 65535)
+; SSE2-NEXT:    [[TMP11:%.*]] = and <2 x i64> [[TMP6]], splat (i64 65535)
+; SSE2-NEXT:    [[TMP12:%.*]] = lshr <2 x i32> [[TMP2]], splat (i32 16)
+; SSE2-NEXT:    [[TMP13:%.*]] = lshr <2 x i32> [[TMP7]], splat (i32 16)
+; SSE2-NEXT:    [[TMP14:%.*]] = add nuw nsw <2 x i64> [[TMP10]], splat (i64 1)
+; SSE2-NEXT:    [[TMP15:%.*]] = add nuw nsw <2 x i64> [[TMP14]], [[TMP11]]
+; SSE2-NEXT:    [[TMP16:%.*]] = lshr <2 x i64> [[TMP15]], splat (i64 1)
+; SSE2-NEXT:    [[TMP17:%.*]] = add nuw nsw <2 x i32> [[TMP12]], splat (i32 1)
+; SSE2-NEXT:    [[TMP18:%.*]] = add nuw nsw <2 x i32> [[TMP17]], [[TMP13]]
+; SSE2-NEXT:    [[TMP19:%.*]] = lshr <2 x i32> [[TMP18]], splat (i32 1)
+; SSE2-NEXT:    [[TMP20:%.*]] = and <2 x i64> [[TMP3]], splat (i64 65535)
+; SSE2-NEXT:    [[TMP21:%.*]] = and <2 x i64> [[TMP8]], splat (i64 65535)
+; SSE2-NEXT:    [[TMP22:%.*]] = add nuw nsw <2 x i64> [[TMP20]], splat (i64 1)
+; SSE2-NEXT:    [[TMP23:%.*]] = add nuw nsw <2 x i64> [[TMP22]], [[TMP21]]
+; SSE2-NEXT:    [[TMP24:%.*]] = add nuw nsw <2 x i64> [[TMP4]], splat (i64 1)
+; SSE2-NEXT:    [[TMP25:%.*]] = add nuw nsw <2 x i64> [[TMP24]], [[TMP9]]
+; SSE2-NEXT:    [[TMP26:%.*]] = shl nuw <2 x i64> [[TMP25]], splat (i64 47)
+; SSE2-NEXT:    [[TMP27:%.*]] = and <2 x i64> [[TMP26]], splat (i64 -281474976710656)
+; SSE2-NEXT:    [[TMP28:%.*]] = shl nuw nsw <2 x i64> [[TMP23]], splat (i64 31)
+; SSE2-NEXT:    [[TMP29:%.*]] = and <2 x i64> [[TMP28]], splat (i64 562945658454016)
+; SSE2-NEXT:    [[TMP30:%.*]] = or <2 x i64> [[TMP27]], [[TMP29]]
+; SSE2-NEXT:    [[TMP31:%.*]] = zext nneg <2 x i32> [[TMP19]] to <2 x i64>
+; SSE2-NEXT:    [[TMP32:%.*]] = shl nuw nsw <2 x i64> [[TMP31]], splat (i64 16)
+; SSE2-NEXT:    [[TMP33:%.*]] = or <2 x i64> [[TMP30]], [[TMP32]]
+; SSE2-NEXT:    [[TMP34:%.*]] = or <2 x i64> [[TMP33]], [[TMP16]]
+; SSE2-NEXT:    [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP34]], i64 0
+; SSE2-NEXT:    [[TMP35:%.*]] = insertvalue { i64, i64 } poison, i64 [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT]], 0
+; SSE2-NEXT:    [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP34]], i64 1
+; SSE2-NEXT:    [[VEC2STRUCT4:%.*]] = insertvalue { i64, i64 } [[TMP35]], i64 [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT]], 1
+; SSE2-NEXT:    ret { i64, i64 } [[VEC2STRUCT4]]
+;
+; SSE4-LABEL: @avgr_8_u16(
+; SSE4-NEXT:  entry:
+; SSE4-NEXT:    [[TMP0:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
+; SSE4-NEXT:    [[TMP1:%.*]] = insertelement <2 x i64> [[TMP0]], i64 [[A_COERCE1:%.*]], i64 1
+; SSE4-NEXT:    [[TMP2:%.*]] = trunc <2 x i64> [[TMP1]] to <2 x i32>
+; SSE4-NEXT:    [[TMP3:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 32)
+; SSE4-NEXT:    [[TMP4:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 48)
+; SSE4-NEXT:    [[TMP5:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
+; SSE4-NEXT:    [[TMP6:%.*]] = insertelement <2 x i64> [[TMP5]], i64 [[B_COERCE1:%.*]], i64 1
+; SSE4-NEXT:    [[TMP7:%.*]] = trunc <2 x i64> [[TMP6]] to <2 x i32>
+; SSE4-NEXT:    [[TMP8:%.*]] = lshr <2 x i64> [[TMP6]], splat (i64 32)
+; SSE4-NEXT:    [[TMP9:%.*]] = lshr <2 x i64> [[TMP6]], splat (i64 48)
+; SSE4-NEXT:    [[TMP10:%.*]] = and <2 x i64> [[TMP1]], splat (i64 65535)
+; SSE4-NEXT:    [[TMP11:%.*]] = and <2 x i64> [[TMP6]], splat (i64 65535)
+; SSE4-NEXT:    [[TMP12:%.*]] = lshr <2 x i32> [[TMP2]], splat (i32 16)
+; SSE4-NEXT:    [[TMP13:%.*]] = lshr <2 x i32> [[TMP7]], splat (i32 16)
+; SSE4-NEXT:    [[TMP14:%.*]] = add nuw nsw <2 x i64> [[TMP10]], splat (i64 1)
+; SSE4-NEXT:    [[TMP15:%.*]] = add nuw nsw <2 x i64> [[TMP14]], [[TMP11]]
+; SSE4-NEXT:    [[TMP16:%.*]] = lshr <2 x i64> [[TMP15]], splat (i64 1)
+; SSE4-NEXT:    [[TMP17:%.*]] = add nuw nsw <2 x i32> [[TMP12]], splat (i32 1)
+; SSE4-NEXT:    [[TMP18:%.*]] = add nuw nsw <2 x i32> [[TMP17]], [[TMP13]]
+; SSE4-NEXT:    [[TMP19:%.*]] = lshr <2 x i32> [[TMP18]], splat (i32 1)
+; SSE4-NEXT:    [[TMP20:%.*]] = and <2 x i64> [[TMP3]], splat (i64 65535)
+; SSE4-NEXT:    [[TMP21:%.*]] = and <2 x i64> [[TMP8]], splat (i64 65535)
+; SSE4-NEXT:    [[TMP22:%.*]] = add nuw nsw <2 x i64> [[TMP20]], splat (i64 1)
+; SSE4-NEXT:    [[TMP23:%.*]] = add nuw nsw <2 x i64> [[TMP22]], [[TMP21]]
+; SSE4-NEXT:    [[TMP24:%.*]] = add nuw nsw <2 x i64> [[TMP4]], splat (i64 1)
+; SSE4-NEXT:    [[TMP25:%.*]] = add nuw nsw <2 x i64> [[TMP24]], [[TMP9]]
+; SSE4-NEXT:    [[TMP26:%.*]] = shl nuw <2 x i64> [[TMP25]], splat (i64 47)
+; SSE4-NEXT:    [[TMP27:%.*]] = and <2 x i64> [[TMP26]], splat (i64 -281474976710656)
+; SSE4-NEXT:    [[TMP28:%.*]] = shl nuw nsw <2 x i64> [[TMP23]], splat (i64 31)
+; SSE4-NEXT:    [[TMP29:%.*]] = and <2 x i64> [[TMP28]], splat (i64 562945658454016)
+; SSE4-NEXT:    [[TMP30:%.*]] = or <2 x i64> [[TMP27]], [[TMP29]]
+; SSE4-NEXT:    [[TMP31:%.*]] = zext nneg <2 x i32> [[TMP19]] to <2 x i64>
+; SSE4-NEXT:    [[TMP32:%.*]] = shl nuw nsw <2 x i64> [[TMP31]], splat (i64 16)
+; SSE4-NEXT:    [[TMP33:%.*]] = or <2 x i64> [[TMP30]], [[TMP32]]
+; SSE4-NEXT:    [[TMP34:%.*]] = or <2 x i64> [[TMP33]], [[TMP16]]
+; SSE4-NEXT:    [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP34]], i64 0
+; SSE4-NEXT:    [[TMP35:%.*]] = insertvalue { i64, i64 } poison, i64 [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT]], 0
+; SSE4-NEXT:    [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP34]], i64 1
+; SSE4-NEXT:    [[VEC2STRUCT4:%.*]] = insertvalue { i64, i64 } [[TMP35]], i64 [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT]], 1
+; SSE4-NEXT:    ret { i64, i64 } [[VEC2STRUCT4]]
+;
+; AVX2-LABEL: @avgr_8_u16(
+; AVX2-NEXT:  entry:
+; AVX2-NEXT:    [[TMP0:%.*]] = insertelement <2 x i64> poison, i64 [[A_COERCE0:%.*]], i64 0
+; AVX2-NEXT:    [[TMP1:%.*]] = insertelement <2 x i64> [[TMP0]], i64 [[A_COERCE1:%.*]], i64 1
+; AVX2-NEXT:    [[TMP2:%.*]] = trunc <2 x i64> [[TMP1]] to <2 x i32>
+; AVX2-NEXT:    [[TMP3:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 32)
+; AVX2-NEXT:    [[TMP4:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 48)
+; AVX2-NEXT:    [[TMP5:%.*]] = insertelement <2 x i64> poison, i64 [[B_COERCE0:%.*]], i64 0
+; AVX2-NEXT:    [[TMP6:%.*]] = insertelement <2 x i64> [[TMP5]], i64 [[B_COERCE1:%.*]], i64 1
+; AVX2-NEXT:    [[TMP7:%.*]] = trunc <2 x i64> [[TMP6]] to <2 x i32>
+; AVX2-NEXT:    [[TMP8:%.*]] = lshr <2 x i64> [[TMP6]], splat (i64 32)
+; AVX2-NEXT:    [[TMP9:%.*]] = lshr <2 x i64> [[TMP6]], splat (i64 48)
+; AVX2-NEXT:    [[TMP10:%.*]] = and <2 x i64> [[TMP1]], splat (i64 65535)
+; AVX2-NEXT:    [[TMP11:%.*]] = and <2 x i64> [[TMP6]], splat (i64 65535)
+; AVX2-NEXT:    [[TMP12:%.*]] = lshr <2 x i32> [[TMP2]], splat (i32 16)
+; AVX2-NEXT:    [[TMP13:%.*]] = lshr <2 x i32> [[TMP7]], splat (i32 16)
+; AVX2-NEXT:    [[TMP14:%.*]] = add nuw nsw <2 x i64> [[TMP10]], splat (i64 1)
+; AVX2-NEXT:    [[TMP15:%.*]] = add nuw nsw <2 x i64> [[TMP14]], [[TMP11]]
+; AVX2-NEXT:    [[TMP16:%.*]] = lshr <2 x i64> [[TMP15]], splat (i64 1)
+; AVX2-NEXT:    [[TMP17:%.*]] = add nuw nsw <2 x i32> [[TMP12]], splat (i32 1)
+; AVX2-NEXT:    [[TMP18:%.*]] = add nuw nsw <2 x i32> [[TMP17]], [[TMP13]]
+; AVX2-NEXT:    [[TMP19:%.*]] = lshr <2 x i32> [[TMP18]], splat (i32 1)
+; AVX2-NEXT:    [[TMP20:%.*]] = and <2 x i64> [[TMP3]], splat (i64 65535)
+; AVX2-NEXT:    [[TMP21:%.*]] = and <2 x i64> [[TMP8]], splat (i64 65535)
+; AVX2-NEXT:    [[TMP22:%.*]] = add nuw nsw <2 x i64> [[TMP20]], splat (i64 1)
+; AVX2-NEXT:    [[TMP23:%.*]] = add nuw nsw <2 x i64> [[TMP22]], [[TMP21]]
+; AVX2-NEXT:    [[TMP24:%.*]] = add nuw nsw <2 x i64> [[TMP4]], splat (i64 1)
+; AVX2-NEXT:    [[TMP25:%.*]] = add nuw nsw <2 x i64> [[TMP24]], [[TMP9]]
+; AVX2-NEXT:    [[TMP26:%.*]] = shl nuw <2 x i64> [[TMP25]], splat (i64 47)
+; AVX2-NEXT:    [[TMP27:%.*]] = and <2 x i64> [[TMP26]], splat (i64 -281474976710656)
+; AVX2-NEXT:    [[TMP28:%.*]] = shl nuw nsw <2 x i64> [[TMP23]], splat (i64 31)
+; AVX2-NEXT:    [[TMP29:%.*]] = and <2 x i64> [[TMP28]], splat (i64 562945658454016)
+; AVX2-NEXT:    [[TMP30:%.*]] = or <2 x i64> [[TMP27]], [[TMP29]]
+; AVX2-NEXT:    [[TMP31:%.*]] = zext nneg <2 x i32> [[TMP19]] to <2 x i64>
+; AVX2-NEXT:    [[TMP32:%.*]] = shl nuw nsw <2 x i64> [[TMP31]], splat (i64 16)
+; AVX2-NEXT:    [[TMP33:%.*]] = or <2 x i64> [[TMP30]], [[TMP32]]
+; AVX2-NEXT:    [[TMP34:%.*]] = or <2 x i64> [[TMP33]], [[TMP16]]
+; AVX2-NEXT:    [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP34]], i64 0
+; AVX2-NEXT:    [[TMP35:%.*]] = insertvalue { i64, i64 } poison, i64 [[VEC2STRUCT_SLOT_SROA_0_0_VEC_EXTRACT]], 0
+; AVX2-NEXT:    [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT:%.*]] = extractelement <2 x i64> [[TMP34]], i64 1
+; AVX2-NEXT:    [[VEC2STRUCT4:%.*]] = insertvalue { i64, i64 } [[TMP35]], i64 [[VEC2STRUCT_SLOT_SROA_0_8_VEC_EXTRACT]], 1
+; AVX2-NEXT:    ret { i64, i64 } [[VEC2STRUCT4]]
+;
+; AVX512-LABEL: @avgr_8_u16(
+; AVX512-NEXT:  entry:
+; AVX512-NEXT:    [[TMP0:%.*]] = trunc i64 [[A_COERCE0:%.*]] to i32
+; AVX512-NEXT:    [[TMP1:%.*]] = insertelement <2 x i32> poison, i32 [[TMP0]], i64 0
+; AVX512-NEXT:    [[TMP2:%.*]] = trunc i64 [[A_COERCE1:%.*]] to i32
+; AVX512-NEXT:    [[TMP3:%.*]] = insertelement <2 x i32> [[TMP1]], i32 [[TMP2]], i64 1
+; AVX512-NEXT:    [[TMP4:%.*]] = trunc i64 [[B_COERCE0:%.*]] to i32
+; AVX512-NEXT:    [[TMP5:%.*]] = insertelement <2 x i32> poison, i32 [[TMP4]], i64 0
+; AVX512-NEXT:    [[TMP6:%.*]] = trunc i64 [[B_COERCE1:%.*]] to i32
+; AVX512-NEXT:    [[TMP7:%.*]] = insertelement <2 x i32> [[TMP5]], i32 [[TMP6]], i64 1
+; AVX512-NEXT:    [[TMP8:%.*]] = insertelement <4 x i64> <i64 poison, i64 -1, i64 poison, i64 poison>, i64 [[B_COERCE0]], i64 0
+; AVX512-NEXT:    [[TMP9:%.*]] = shufflevector <4 x i64> [[TMP8]], <4 x i64> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
+; AVX512-NEXT:    [[TMP10:%.*]] = lshr <4 x i64> [[TMP9]], <i64 48, i64 32, i64 0, i64 0>
+; AVX512-NEXT:    [[TMP11:%.*]] = insertelement <4 x i64> poison, i64 [[A_COERCE0]], i64 0
+; AVX512-NEXT:    [[TMP12:%.*]] = and <4 x i64> [[TMP10]], <i64 -1, i64 65535, i64 65535, i64 0>
+; AVX512-NEXT:    [[TMP13:%.*]] = insertelement <4 x i64> <i64 poison, i64 -1, i64 poison, i64 poison>, i64 [[B_COERCE1]], i64 0
+; AVX512-NEXT:    [[TMP14:%.*]] = shufflevector <4 x i64> [[TMP13]], <4 x i64> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
+; AVX512-NEXT:    [[TMP15:%.*]] = lshr <4 x i64> [[TMP14]], <i64 48, i64 32, i64 0, i64 0>
+; AVX512-NEXT:    [[TMP16:%.*]] = lshr <2 x i32> [[TMP3]], splat (i32 16)
+; AVX512-NEXT:    [[TMP17:%.*]] = lshr <2 x i32> [[TMP7]], splat (i32 16)
+; AVX512-NEXT:    [[TMP18:%.*]] = add nuw nsw <2 x i32> [[TMP16]], splat (i32 1)
+; AVX512-NEXT:    [[TMP19:%.*]] = add nuw nsw <2 x i32> [[TMP18]], [[TMP17]]
+; AVX512-NEXT:    [[TMP20:%.*]] = lshr <2 x i32> [[TMP19]], splat (i32 1)
+; AVX512-NEXT:    [[TMP21:%.*]] = shufflevector <2 x i32> [[TMP20]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; AVX512-NEXT:    [[TMP22:%.*]] = zext <4 x i32> [[TMP21]] to <4 x i64>
+; AVX512-NEXT:    [[TMP23:%.*]] = shufflevector <4 x i64> [[TMP11]], <4 x i64> [[TMP22]], <4 x i32> <i32 0, i32 0, i32 0, i32 4>
+; AVX512-NEXT:    [[TMP24:%.*]] = lshr <4 x i64> [[TMP23]], <i64 48, i64 32, i64 0, i64 0>
+; AVX512-NEXT:    [[TMP25:%.*]] = and <4 x i64> [[TMP24]], <i64 -1, i64 65535, i64 65535, i64 -1>
+; AVX512-NEXT:    [[TMP26:%.*]] = add nuw nsw <4 x i64> [[TMP25]], <i64 1, i64 1, i64 1, i64 0>
+; AVX512-NEXT:    [[TMP27:%.*]] = add nuw nsw <4 x i64> [[TMP26]], [[TMP12]]
+; AVX512-NEXT:    [[TMP28:%.*]] = shl nuw <4 x i64> [[TMP27]], <i64 47, i64 31, i64 1, i64 16>
+; AVX512-NEXT:    [[TMP29:%.*]] = lshr <4 x i64> [[TMP27]], <i64 47, i64 31, i64 1, i64 16>
+; AVX512-NEXT:    [[TMP30:%.*]] = shufflevector <4 x i64> [[TMP28]], <4 x i64> [[TMP29]], <4 x i32> <i32 0, i32 1, i32 6, i32 3>
+; AVX512-NEXT:    [[TMP31:%.*]] = and <4 x i64> [[TMP30]], <i64 -281474976710656, i64 562945658454016, i64 -1, i64 -1>
+; AVX512-NEXT:    [[TMP32:%.*]] = tail call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP31]])
+; AVX512-NEXT:    [[DOTFCA_0_INSERT:%.*]] = insertvalue { i64, i64 } poison, i64 [[TMP32]], 0
+; AVX512-NEXT:    [[TMP33:%.*]] = insertelement <4 x i64> poison, i64 [[A_COERCE1]], i64 0
+; AVX512-NEXT:    [[TMP34:%.*]] = shufflevector <4 x i64> [[TMP33]], <4 x i64> [[TMP22]], <4 x i32> <i32 0, i32 0, i32 0, i32 5>
+; AVX512-NEXT:    [[TMP35:%.*]] = lshr <4 x i64> [[TMP34]], <i64 48, i64 32, i64 0, i64 0>
+; AVX512-NEXT:    [[TMP36:%.*]] = and <4 x i64> [[TMP35]], <i64 -1, i64 65535, i64 65535, i64 -1>
+; AVX512-NEXT:    [[TMP37:%.*]] = and <4 x i64> [[TMP15]], <i64 -1, i64 65535, i64 65535, i64 0>
+; AVX512-NEXT:    [[TMP38:%.*]] = add nuw nsw <4 x i64> [[TMP36]], <i64 1, i64 1, i64 1, i64 0>
+; AVX512-NEXT:    [[TMP39:%.*]] = add nuw nsw <4 x i64> [[TMP38]], [[TMP37]]
+; AVX512-NEXT:    [[TMP40:%.*]] = shl nuw <4 x i64> [[TMP39]], <i64 47, i64 31, i64 1, i64 16>
+; AVX512-NEXT:    [[TMP41:%.*]] = lshr <4 x i64> [[TMP39]], <i64 47, i64 31, i64 1, i64 16>
+; AVX512-NEXT:    [[TMP42:%.*]] = shufflevector <4 x i64> [[TMP40]], <4 x i64> [[TMP41]], <4 x i32> <i32 0, i32 1, i32 6, i32 3>
+; AVX512-NEXT:    [[TMP43:%.*]] = and <4 x i64> [[TMP42]], <i64 -281474976710656, i64 562945658454016, i64 -1, i64 -1>
+; AVX512-NEXT:    [[TMP44:%.*]] = tail call i64 @llvm.vector.reduce.or.v4i64(<4 x i64> [[TMP43]])
+; AVX512-NEXT:    [[DOTFCA_1_INSERT:%.*]] = insertvalue { i64, i64 } [[DOTFCA_0_INSERT]], i64 [[TMP44]], 1
+; AVX512-NEXT:    ret { i64, i64 } [[DOTFCA_1_INSERT]]
 ;
 entry:
   %retval = alloca %"struct.std::array8", align 2
@@ -1033,3 +1148,5 @@ for.body:                                         ; preds = %for.cond
   %inc = add nuw nsw i64 %i.0, 1
   br label %for.cond
 }
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK: {{.*}}



More information about the llvm-commits mailing list