[llvm] [X86] Enable MaximizeBandwidth by default and price predicate mask-expansion fanout (PR #201666)
Sumukh J Bharadwaj via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 10 07:08:39 PDT 2026
https://github.com/amd-subharad updated https://github.com/llvm/llvm-project/pull/201666
>From 5dbaab2754a70b91465e037a605ba51278699937 Mon Sep 17 00:00:00 2001
From: Sumukh Bharadwaj <Sumukh.Bharadwaj at amd.com>
Date: Tue, 18 Aug 2026 03:35:20 +0530
Subject: [PATCH 1/2] [X86][LoopVectorize] Enable MaximizeBandwidth by default
on X86
Enable the MaximizeBandwidth heuristic by default for X86 fixed-width
vectors via shouldMaximizeVectorBandwidth(), matching the AArch64 Neon
precedent, so the loop vectorizer can select a VF from the smallest
element type in mixed-width loops.
Existing LoopVectorize/X86 and PhaseOrdering/X86 tests whose VF changes
under the new default have their CHECK lines regenerated with MaxBW
enabled, so they validate the new behavior directly rather than being
opted out with -vectorizer-maximize-bandwidth=false. A focused
maxbw-cast-cost.ll pins the VF-from-smallest-type selection.
Enabling the heuristic is neutral-to-positive in benchmarking. The one
mispricing it exposes -- the X86 per-part cost tables under-counting
predicate mask-expansion fanout, which let masked / gather-scatter loops
over-widen -- is addressed by the companion X86 TTI cost-model commit in
this series, so the fix stays in the cost model rather than a generic
LV-side floor or vectorizer-level workaround.
---
llvm/lib/Target/X86/X86TargetTransformInfo.h | 6 +
.../X86/conditional-scalar-assignment.ll | 109 ++-
.../X86/cost-conditional-branches.ll | 40 +-
.../LoopVectorize/X86/cost-model.ll | 26 +-
.../LoopVectorize/X86/gcc-examples.ll | 144 ++-
.../LoopVectorize/X86/induction-costs.ll | 60 +-
.../LoopVectorize/X86/masked_load_store.ll | 883 +++++++-----------
.../LoopVectorize/X86/maxbw-cast-cost.ll | 353 +++++++
.../Transforms/LoopVectorize/X86/no_fpmath.ll | 2 +-
.../X86/no_fpmath_with_hotness.ll | 2 +-
.../X86/nondetermisitic-widening-cost.ll | 48 +-
.../Transforms/LoopVectorize/X86/pr47437.ll | 85 +-
.../LoopVectorize/X86/reduction-crash.ll | 24 +-
.../X86/replicating-load-store-costs.ll | 303 +++---
.../LoopVectorize/X86/strided_load_cost.ll | 124 ++-
.../X86/vector_ptr_load_store.ll | 4 +-
.../X86/vectorization-remarks-loopid-dbg.ll | 149 ++-
.../X86/vectorization-remarks.ll | 171 +++-
.../PhaseOrdering/X86/pixel-splat.ll | 55 +-
.../X86/preserve-access-group.ll | 28 +-
.../X86/vector-reduction-known-first-value.ll | 274 +++++-
21 files changed, 1974 insertions(+), 916 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/X86/maxbw-cast-cost.ll
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.h b/llvm/lib/Target/X86/X86TargetTransformInfo.h
index 422e4aad2316c..4811e6afdaa46 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.h
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.h
@@ -64,6 +64,12 @@ class X86TTIImpl final : public BasicTTIImplBase<X86TTIImpl> {
TypeSize
getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override;
unsigned getLoadStoreVecRegBitWidth(unsigned AS) const override;
+ bool shouldMaximizeVectorBandwidth(
+ TargetTransformInfo::RegisterKind K) const override {
+ assert(K != TargetTransformInfo::RGK_Scalar &&
+ "Expected vector register kind");
+ return K == TargetTransformInfo::RGK_FixedWidthVector;
+ }
unsigned getMaxInterleaveFactor(ElementCount VF,
bool HasUnorderedReductions) const override;
InstructionCost getArithmeticInstrCost(
diff --git a/llvm/test/Transforms/LoopVectorize/X86/conditional-scalar-assignment.ll b/llvm/test/Transforms/LoopVectorize/X86/conditional-scalar-assignment.ll
index f6bd69acb775b..ec3bfaecf4941 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/conditional-scalar-assignment.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/conditional-scalar-assignment.ll
@@ -388,22 +388,59 @@ define i32 @multi_use_cmp_for_csa_int_select(i64 %N, ptr %data, i32 %a) {
; X86-LABEL: define i32 @multi_use_cmp_for_csa_int_select(
; X86-SAME: i64 [[N:%.*]], ptr [[DATA:%.*]], i32 [[A:%.*]]) {
; X86-NEXT: [[ENTRY:.*]]:
+; X86-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; X86-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; X86: [[VECTOR_PH]]:
+; X86-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
+; X86-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; X86-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[A]], i64 0
+; X86-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
; X86-NEXT: br label %[[LOOP:.*]]
; X86: [[LOOP]]:
-; X86-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; X86-NEXT: [[DATA_PHI:%.*]] = phi i32 [ -1, %[[ENTRY]] ], [ [[SELECT_DATA:%.*]], %[[LOOP]] ]
-; X86-NEXT: [[IDX_PHI:%.*]] = phi i64 [ -1, %[[ENTRY]] ], [ [[SELECT_IDX:%.*]], %[[LOOP]] ]
+; X86-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP]] ]
+; X86-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[LOOP]] ]
+; X86-NEXT: [[VEC_PHI:%.*]] = phi <4 x i32> [ splat (i32 -1), %[[VECTOR_PH]] ], [ [[TMP7:%.*]], %[[LOOP]] ]
+; X86-NEXT: [[TMP1:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP6:%.*]], %[[LOOP]] ]
+; X86-NEXT: [[VEC_PHI1:%.*]] = phi <4 x i64> [ splat (i64 -9223372036854775808), %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[LOOP]] ]
; X86-NEXT: [[LD_ADDR:%.*]] = getelementptr inbounds i32, ptr [[DATA]], i64 [[IV]]
-; X86-NEXT: [[LD:%.*]] = load i32, ptr [[LD_ADDR]], align 4
+; X86-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[LD_ADDR]], align 4
+; X86-NEXT: [[TMP3:%.*]] = icmp slt <4 x i32> [[BROADCAST_SPLAT]], [[WIDE_LOAD]]
+; X86-NEXT: [[TMP4:%.*]] = freeze <4 x i1> [[TMP3]]
+; X86-NEXT: [[TMP5:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP4]])
+; X86-NEXT: [[TMP6]] = select i1 [[TMP5]], <4 x i1> [[TMP3]], <4 x i1> [[TMP1]]
+; X86-NEXT: [[TMP7]] = select i1 [[TMP5]], <4 x i32> [[WIDE_LOAD]], <4 x i32> [[VEC_PHI]]
+; X86-NEXT: [[TMP8]] = select <4 x i1> [[TMP3]], <4 x i64> [[VEC_IND]], <4 x i64> [[VEC_PHI1]]
+; X86-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
+; X86-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; X86-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; X86-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP6:![0-9]+]]
+; X86: [[MIDDLE_BLOCK]]:
+; X86-NEXT: [[TMP10:%.*]] = call i32 @llvm.experimental.vector.extract.last.active.v4i32(<4 x i32> [[TMP7]], <4 x i1> [[TMP6]], i32 -1)
+; X86-NEXT: [[TMP11:%.*]] = call i64 @llvm.vector.reduce.smax.v4i64(<4 x i64> [[TMP8]])
+; X86-NEXT: [[TMP12:%.*]] = icmp ne i64 [[TMP11]], -9223372036854775808
+; X86-NEXT: [[TMP13:%.*]] = select i1 [[TMP12]], i64 [[TMP11]], i64 -1
+; X86-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; X86-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; X86: [[SCALAR_PH]]:
+; X86-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; X86-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP10]], %[[MIDDLE_BLOCK]] ], [ -1, %[[ENTRY]] ]
+; X86-NEXT: [[BC_MERGE_RDX2:%.*]] = phi i64 [ [[TMP13]], %[[MIDDLE_BLOCK]] ], [ -1, %[[ENTRY]] ]
+; X86-NEXT: br label %[[LOOP1:.*]]
+; X86: [[LOOP1]]:
+; X86-NEXT: [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; X86-NEXT: [[DATA_PHI:%.*]] = phi i32 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[SELECT_DATA:%.*]], %[[LOOP1]] ]
+; X86-NEXT: [[IDX_PHI:%.*]] = phi i64 [ [[BC_MERGE_RDX2]], %[[SCALAR_PH]] ], [ [[SELECT_IDX:%.*]], %[[LOOP1]] ]
+; X86-NEXT: [[LD_ADDR1:%.*]] = getelementptr inbounds i32, ptr [[DATA]], i64 [[IV1]]
+; X86-NEXT: [[LD:%.*]] = load i32, ptr [[LD_ADDR1]], align 4
; X86-NEXT: [[SELECT_CMP:%.*]] = icmp slt i32 [[A]], [[LD]]
; X86-NEXT: [[SELECT_DATA]] = select i1 [[SELECT_CMP]], i32 [[LD]], i32 [[DATA_PHI]]
-; X86-NEXT: [[SELECT_IDX]] = select i1 [[SELECT_CMP]], i64 [[IV]], i64 [[IDX_PHI]]
-; X86-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; X86-NEXT: [[SELECT_IDX]] = select i1 [[SELECT_CMP]], i64 [[IV1]], i64 [[IDX_PHI]]
+; X86-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; X86-NEXT: [[EXIT_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT:.*]], label %[[LOOP]]
+; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP7:![0-9]+]]
; X86: [[EXIT]]:
-; X86-NEXT: [[SELECT_DATA_LCSSA:%.*]] = phi i32 [ [[SELECT_DATA]], %[[LOOP]] ]
-; X86-NEXT: [[SELECT_IDX_LCSSA:%.*]] = phi i64 [ [[SELECT_IDX]], %[[LOOP]] ]
+; X86-NEXT: [[SELECT_DATA_LCSSA:%.*]] = phi i32 [ [[SELECT_DATA]], %[[LOOP1]] ], [ [[TMP10]], %[[MIDDLE_BLOCK]] ]
+; X86-NEXT: [[SELECT_IDX_LCSSA:%.*]] = phi i64 [ [[SELECT_IDX]], %[[LOOP1]] ], [ [[TMP13]], %[[MIDDLE_BLOCK]] ]
; X86-NEXT: [[IDX:%.*]] = trunc i64 [[SELECT_IDX_LCSSA]] to i32
; X86-NEXT: [[RES:%.*]] = add i32 [[IDX]], [[SELECT_DATA_LCSSA]]
; X86-NEXT: ret i32 [[RES]]
@@ -411,42 +448,42 @@ define i32 @multi_use_cmp_for_csa_int_select(i64 %N, ptr %data, i32 %a) {
; AVX512-LABEL: define i32 @multi_use_cmp_for_csa_int_select(
; AVX512-SAME: i64 [[N:%.*]], ptr [[DATA:%.*]], i32 [[A:%.*]]) #[[ATTR0]] {
; AVX512-NEXT: [[ENTRY:.*]]:
-; AVX512-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 16
+; AVX512-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 32
; AVX512-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX512: [[VECTOR_PH]]:
-; AVX512-NEXT: [[N_MOD_VF:%.*]] = and i64 [[N]], 7
+; AVX512-NEXT: [[N_MOD_VF:%.*]] = and i64 [[N]], 15
; AVX512-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
-; AVX512-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i32> poison, i32 [[A]], i64 0
-; AVX512-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT]], <8 x i32> poison, <8 x i32> zeroinitializer
+; AVX512-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <16 x i32> poison, i32 [[A]], i64 0
+; AVX512-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <16 x i32> [[BROADCAST_SPLATINSERT]], <16 x i32> poison, <16 x i32> zeroinitializer
; AVX512-NEXT: br label %[[LOOP:.*]]
; AVX512: [[LOOP]]:
; AVX512-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP]] ]
-; AVX512-NEXT: [[VEC_IND:%.*]] = phi <8 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[LOOP]] ]
-; AVX512-NEXT: [[VEC_PHI:%.*]] = phi <8 x i32> [ splat (i32 -1), %[[VECTOR_PH]] ], [ [[TMP6:%.*]], %[[LOOP]] ]
-; AVX512-NEXT: [[TMP0:%.*]] = phi <8 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[LOOP]] ]
-; AVX512-NEXT: [[VEC_PHI1:%.*]] = phi <8 x i64> [ splat (i64 -9223372036854775808), %[[VECTOR_PH]] ], [ [[TMP7:%.*]], %[[LOOP]] ]
+; AVX512-NEXT: [[VEC_IND:%.*]] = phi <16 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9, i64 10, i64 11, i64 12, i64 13, i64 14, i64 15>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[LOOP]] ]
+; AVX512-NEXT: [[VEC_PHI:%.*]] = phi <16 x i32> [ splat (i32 -1), %[[VECTOR_PH]] ], [ [[TMP7:%.*]], %[[LOOP]] ]
+; AVX512-NEXT: [[TMP1:%.*]] = phi <16 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP6:%.*]], %[[LOOP]] ]
+; AVX512-NEXT: [[VEC_PHI1:%.*]] = phi <16 x i64> [ splat (i64 -9223372036854775808), %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[LOOP]] ]
; AVX512-NEXT: [[LD_ADDR:%.*]] = getelementptr inbounds i32, ptr [[DATA]], i64 [[IV]]
-; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[LD_ADDR]], align 4
-; AVX512-NEXT: [[TMP2:%.*]] = icmp slt <8 x i32> [[BROADCAST_SPLAT]], [[WIDE_LOAD]]
-; AVX512-NEXT: [[TMP3:%.*]] = freeze <8 x i1> [[TMP2]]
-; AVX512-NEXT: [[TMP4:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP3]])
-; AVX512-NEXT: [[TMP5]] = select i1 [[TMP4]], <8 x i1> [[TMP2]], <8 x i1> [[TMP0]]
-; AVX512-NEXT: [[TMP6]] = select i1 [[TMP4]], <8 x i32> [[WIDE_LOAD]], <8 x i32> [[VEC_PHI]]
-; AVX512-NEXT: [[TMP7]] = select <8 x i1> [[TMP2]], <8 x i64> [[VEC_IND]], <8 x i64> [[VEC_PHI1]]
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 8
-; AVX512-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <8 x i64> [[VEC_IND]], splat (i64 8)
+; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[LD_ADDR]], align 4
+; AVX512-NEXT: [[TMP3:%.*]] = icmp slt <16 x i32> [[BROADCAST_SPLAT]], [[WIDE_LOAD]]
+; AVX512-NEXT: [[TMP4:%.*]] = freeze <16 x i1> [[TMP3]]
+; AVX512-NEXT: [[TMP5:%.*]] = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> [[TMP4]])
+; AVX512-NEXT: [[TMP6]] = select i1 [[TMP5]], <16 x i1> [[TMP3]], <16 x i1> [[TMP1]]
+; AVX512-NEXT: [[TMP7]] = select i1 [[TMP5]], <16 x i32> [[WIDE_LOAD]], <16 x i32> [[VEC_PHI]]
+; AVX512-NEXT: [[TMP9]] = select <16 x i1> [[TMP3]], <16 x i64> [[VEC_IND]], <16 x i64> [[VEC_PHI1]]
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 16
+; AVX512-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <16 x i64> [[VEC_IND]], splat (i64 16)
; AVX512-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX512-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
-; AVX512-NEXT: [[TMP9:%.*]] = call i32 @llvm.experimental.vector.extract.last.active.v8i32(<8 x i32> [[TMP6]], <8 x i1> [[TMP5]], i32 -1)
-; AVX512-NEXT: [[TMP10:%.*]] = call i64 @llvm.vector.reduce.smax.v8i64(<8 x i64> [[TMP7]])
+; AVX512-NEXT: [[TMP13:%.*]] = call i32 @llvm.experimental.vector.extract.last.active.v16i32(<16 x i32> [[TMP7]], <16 x i1> [[TMP6]], i32 -1)
+; AVX512-NEXT: [[TMP10:%.*]] = call i64 @llvm.vector.reduce.smax.v16i64(<16 x i64> [[TMP9]])
; AVX512-NEXT: [[TMP11:%.*]] = icmp ne i64 [[TMP10]], -9223372036854775808
; AVX512-NEXT: [[TMP12:%.*]] = select i1 [[TMP11]], i64 [[TMP10]], i64 -1
; AVX512-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; AVX512-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
; AVX512: [[SCALAR_PH]]:
; AVX512-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
-; AVX512-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP9]], %[[MIDDLE_BLOCK]] ], [ -1, %[[ENTRY]] ]
+; AVX512-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP13]], %[[MIDDLE_BLOCK]] ], [ -1, %[[ENTRY]] ]
; AVX512-NEXT: [[BC_MERGE_RDX2:%.*]] = phi i64 [ [[TMP12]], %[[MIDDLE_BLOCK]] ], [ -1, %[[ENTRY]] ]
; AVX512-NEXT: br label %[[LOOP1:.*]]
; AVX512: [[LOOP1]]:
@@ -462,7 +499,7 @@ define i32 @multi_use_cmp_for_csa_int_select(i64 %N, ptr %data, i32 %a) {
; AVX512-NEXT: [[EXIT_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
; AVX512-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP9:![0-9]+]]
; AVX512: [[EXIT]]:
-; AVX512-NEXT: [[SELECT_DATA_LCSSA:%.*]] = phi i32 [ [[SELECT_DATA]], %[[LOOP1]] ], [ [[TMP9]], %[[MIDDLE_BLOCK]] ]
+; AVX512-NEXT: [[SELECT_DATA_LCSSA:%.*]] = phi i32 [ [[SELECT_DATA]], %[[LOOP1]] ], [ [[TMP13]], %[[MIDDLE_BLOCK]] ]
; AVX512-NEXT: [[SELECT_IDX_LCSSA:%.*]] = phi i64 [ [[SELECT_IDX]], %[[LOOP1]] ], [ [[TMP12]], %[[MIDDLE_BLOCK]] ]
; AVX512-NEXT: [[IDX:%.*]] = trunc i64 [[SELECT_IDX_LCSSA]] to i32
; AVX512-NEXT: [[RES:%.*]] = add i32 [[IDX]], [[SELECT_DATA_LCSSA]]
@@ -652,7 +689,7 @@ define i32 @int_select_with_extra_arith_payload(i64 %N, ptr readonly %A, ptr rea
; X86-NEXT: [[TMP11]] = select i1 [[TMP9]], <4 x i32> [[WIDE_LOAD]], <4 x i32> [[VEC_PHI]]
; X86-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
; X86-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; X86-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP6:![0-9]+]]
+; X86-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
; X86: [[MIDDLE_BLOCK]]:
; X86-NEXT: [[TMP13:%.*]] = call i32 @llvm.experimental.vector.extract.last.active.v4i32(<4 x i32> [[TMP11]], <4 x i1> [[TMP10]], i32 -1)
; X86-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
@@ -677,7 +714,7 @@ define i32 @int_select_with_extra_arith_payload(i64 %N, ptr readonly %A, ptr rea
; X86-NEXT: [[SELECT_A]] = select i1 [[SELECT_CMP]], i32 [[LD_A]], i32 [[A_PHI]]
; X86-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; X86-NEXT: [[EXIT_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP7:![0-9]+]]
+; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP9:![0-9]+]]
; X86: [[EXIT]]:
; X86-NEXT: [[SELECT_A_LCSSA:%.*]] = phi i32 [ [[SELECT_A]], %[[LOOP1]] ], [ [[TMP13]], %[[MIDDLE_BLOCK]] ]
; X86-NEXT: ret i32 [[SELECT_A_LCSSA]]
@@ -793,7 +830,7 @@ define i8 @simple_csa_byte_select(i64 %N, ptr %data, i8 %a) {
; X86-NEXT: [[TMP6]] = select i1 [[TMP4]], <16 x i8> [[WIDE_LOAD]], <16 x i8> [[VEC_PHI]]
; X86-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 16
; X86-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; X86-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
+; X86-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP10:![0-9]+]]
; X86: [[MIDDLE_BLOCK]]:
; X86-NEXT: [[TMP8:%.*]] = call i8 @llvm.experimental.vector.extract.last.active.v16i8(<16 x i8> [[TMP6]], <16 x i1> [[TMP5]], i8 -1)
; X86-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
@@ -811,7 +848,7 @@ define i8 @simple_csa_byte_select(i64 %N, ptr %data, i8 %a) {
; X86-NEXT: [[SELECT_DATA]] = select i1 [[SELECT_CMP]], i8 [[LD]], i8 [[DATA_PHI]]
; X86-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; X86-NEXT: [[EXIT_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP9:![0-9]+]]
+; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP11:![0-9]+]]
; X86: [[EXIT]]:
; X86-NEXT: [[SELECT_DATA_LCSSA:%.*]] = phi i8 [ [[SELECT_DATA]], %[[LOOP1]] ], [ [[TMP8]], %[[MIDDLE_BLOCK]] ]
; X86-NEXT: ret i8 [[SELECT_DATA_LCSSA]]
@@ -915,7 +952,7 @@ define i32 @simple_csa_int_select_use_interleave(i64 %N, ptr %data, i32 %a) {
; X86-NEXT: [[TMP13]] = select i1 [[TMP4]], <4 x i32> [[WIDE_LOAD2]], <4 x i32> [[VEC_PHI1]]
; X86-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 8
; X86-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; X86-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP10:![0-9]+]]
+; X86-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP12:![0-9]+]]
; X86: [[MIDDLE_BLOCK]]:
; X86-NEXT: [[TMP8:%.*]] = call i32 @llvm.experimental.vector.extract.last.active.v4i32(<4 x i32> [[TMP6]], <4 x i1> [[TMP5]], i32 -1)
; X86-NEXT: [[TMP16:%.*]] = call i32 @llvm.experimental.vector.extract.last.active.v4i32(<4 x i32> [[TMP13]], <4 x i1> [[TMP11]], i32 [[TMP8]])
@@ -934,7 +971,7 @@ define i32 @simple_csa_int_select_use_interleave(i64 %N, ptr %data, i32 %a) {
; X86-NEXT: [[SELECT_DATA]] = select i1 [[SELECT_CMP]], i32 [[LD]], i32 [[DATA_PHI]]
; X86-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; X86-NEXT: [[EXIT_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP11:![0-9]+]]
+; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP13:![0-9]+]]
; X86: [[EXIT]]:
; X86-NEXT: [[SELECT_DATA_LCSSA:%.*]] = phi i32 [ [[SELECT_DATA]], %[[LOOP1]] ], [ [[TMP16]], %[[MIDDLE_BLOCK]] ]
; X86-NEXT: ret i32 [[SELECT_DATA_LCSSA]]
diff --git a/llvm/test/Transforms/LoopVectorize/X86/cost-conditional-branches.ll b/llvm/test/Transforms/LoopVectorize/X86/cost-conditional-branches.ll
index f0c94e0a78629..960ec43364aed 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/cost-conditional-branches.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/cost-conditional-branches.ll
@@ -871,13 +871,13 @@ define i64 @test_predicated_udiv(i32 %d, i1 %c) #2 {
; CHECK-NEXT: iter.check:
; CHECK-NEXT: br i1 false, label [[VEC_EPILOG_SCALAR_PH:%.*]], label [[VECTOR_MAIN_LOOP_ITER_CHECK:%.*]]
; CHECK: vector.main.loop.iter.check:
-; CHECK-NEXT: br i1 false, label [[VEC_EPILOG_PH:%.*]], label [[VECTOR_PH:%.*]]
+; CHECK-NEXT: br i1 false, label [[VEC_EPILOG_PH:%.*]], label [[ENTRY:%.*]]
; CHECK: vector.ph:
; CHECK-NEXT: [[TMP0:%.*]] = xor i1 [[C:%.*]], true
-; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK-NEXT: br label [[LOOP_HEADER:%.*]]
; CHECK: vector.body:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[PRED_UDIV_CONTINUE62:%.*]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <32 x i32> [ <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>, [[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], [[PRED_UDIV_CONTINUE62]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[IV_NEXT:%.*]], [[LOOP_LATCH:%.*]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <32 x i32> [ <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>, [[ENTRY]] ], [ [[VEC_IND_NEXT:%.*]], [[LOOP_LATCH]] ]
; CHECK-NEXT: [[TMP1:%.*]] = call <32 x i32> @llvm.usub.sat.v32i32(<32 x i32> [[VEC_IND]], <32 x i32> splat (i32 1))
; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF:%.*]], label [[PRED_UDIV_CONTINUE:%.*]]
; CHECK: pred.udiv.if:
@@ -886,7 +886,7 @@ define i64 @test_predicated_udiv(i32 %d, i1 %c) #2 {
; CHECK-NEXT: [[TMP4:%.*]] = insertelement <32 x i32> poison, i32 [[TMP3]], i64 0
; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE]]
; CHECK: pred.udiv.continue:
-; CHECK-NEXT: [[TMP5:%.*]] = phi <32 x i32> [ poison, [[VECTOR_BODY]] ], [ [[TMP4]], [[PRED_UDIV_IF]] ]
+; CHECK-NEXT: [[TMP5:%.*]] = phi <32 x i32> [ poison, [[LOOP_HEADER]] ], [ [[TMP4]], [[PRED_UDIV_IF]] ]
; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF1:%.*]], label [[PRED_UDIV_CONTINUE2:%.*]]
; CHECK: pred.udiv.if1:
; CHECK-NEXT: [[TMP6:%.*]] = extractelement <32 x i32> [[TMP1]], i64 1
@@ -1127,18 +1127,18 @@ define i64 @test_predicated_udiv(i32 %d, i1 %c) #2 {
; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE60]]
; CHECK: pred.udiv.continue60:
; CHECK-NEXT: [[TMP125:%.*]] = phi <32 x i32> [ [[TMP121]], [[PRED_UDIV_CONTINUE58]] ], [ [[TMP124]], [[PRED_UDIV_IF59]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF61:%.*]], label [[PRED_UDIV_CONTINUE62]]
+; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF61:%.*]], label [[LOOP_LATCH]]
; CHECK: pred.udiv.if61:
; CHECK-NEXT: [[TMP126:%.*]] = extractelement <32 x i32> [[TMP1]], i64 31
; CHECK-NEXT: [[TMP127:%.*]] = udiv i32 [[TMP126]], [[D]]
; CHECK-NEXT: [[TMP128:%.*]] = insertelement <32 x i32> [[TMP125]], i32 [[TMP127]], i64 31
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE62]]
+; CHECK-NEXT: br label [[LOOP_LATCH]]
; CHECK: pred.udiv.continue62:
; CHECK-NEXT: [[TMP129:%.*]] = phi <32 x i32> [ [[TMP125]], [[PRED_UDIV_CONTINUE60]] ], [ [[TMP128]], [[PRED_UDIV_IF61]] ]
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 32
+; CHECK-NEXT: [[IV_NEXT]] = add nuw i32 [[IV]], 32
; CHECK-NEXT: [[VEC_IND_NEXT]] = add <32 x i32> [[VEC_IND]], splat (i32 32)
-; CHECK-NEXT: [[TMP130:%.*]] = icmp eq i32 [[INDEX_NEXT]], 992
-; CHECK-NEXT: br i1 [[TMP130]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK-NEXT: [[TMP130:%.*]] = icmp eq i32 [[IV_NEXT]], 992
+; CHECK-NEXT: br i1 [[TMP130]], label [[MIDDLE_BLOCK:%.*]], label [[LOOP_HEADER]], !llvm.loop [[LOOP12:![0-9]+]]
; CHECK: middle.block:
; CHECK-NEXT: [[TMP131:%.*]] = zext <32 x i32> [[TMP129]] to <32 x i64>
; CHECK-NEXT: [[PREDPHI:%.*]] = select i1 [[C]], <32 x i64> zeroinitializer, <32 x i64> [[TMP131]]
@@ -1232,22 +1232,22 @@ define i64 @test_predicated_udiv(i32 %d, i1 %c) #2 {
; CHECK-NEXT: br i1 false, label [[EXIT]], label [[VEC_EPILOG_SCALAR_PH]]
; CHECK: vec.epilog.scalar.ph:
; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ 1000, [[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 992, [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[ITER_CHECK:%.*]] ]
-; CHECK-NEXT: br label [[LOOP_HEADER:%.*]]
+; CHECK-NEXT: br label [[LOOP_HEADER1:%.*]]
; CHECK: loop.header:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], [[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], [[LOOP_LATCH:%.*]] ]
-; CHECK-NEXT: br i1 [[C]], label [[LOOP_LATCH]], label [[THEN:%.*]]
+; CHECK-NEXT: [[IV1:%.*]] = phi i32 [ [[BC_RESUME_VAL]], [[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT1:%.*]], [[LOOP_LATCH1:%.*]] ]
+; CHECK-NEXT: br i1 [[C]], label [[LOOP_LATCH1]], label [[THEN:%.*]]
; CHECK: then:
-; CHECK-NEXT: [[CALL:%.*]] = tail call i32 @llvm.usub.sat.i32(i32 [[IV]], i32 1)
+; CHECK-NEXT: [[CALL:%.*]] = tail call i32 @llvm.usub.sat.i32(i32 [[IV1]], i32 1)
; CHECK-NEXT: [[UDIV:%.*]] = udiv i32 [[CALL]], [[D]]
; CHECK-NEXT: [[ZEXT:%.*]] = zext i32 [[UDIV]] to i64
-; CHECK-NEXT: br label [[LOOP_LATCH]]
+; CHECK-NEXT: br label [[LOOP_LATCH1]]
; CHECK: loop.latch:
-; CHECK-NEXT: [[MERGE:%.*]] = phi i64 [ [[ZEXT]], [[THEN]] ], [ 0, [[LOOP_HEADER]] ]
-; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
-; CHECK-NEXT: [[EC:%.*]] = icmp eq i32 [[IV]], 1000
-; CHECK-NEXT: br i1 [[EC]], label [[EXIT]], label [[LOOP_HEADER]], !llvm.loop [[LOOP15:![0-9]+]]
+; CHECK-NEXT: [[MERGE:%.*]] = phi i64 [ [[ZEXT]], [[THEN]] ], [ 0, [[LOOP_HEADER1]] ]
+; CHECK-NEXT: [[IV_NEXT1]] = add i32 [[IV1]], 1
+; CHECK-NEXT: [[EC:%.*]] = icmp eq i32 [[IV1]], 1000
+; CHECK-NEXT: br i1 [[EC]], label [[EXIT]], label [[LOOP_HEADER1]], !llvm.loop [[LOOP15:![0-9]+]]
; CHECK: exit:
-; CHECK-NEXT: [[MERGE_LCSSA:%.*]] = phi i64 [ [[MERGE]], [[LOOP_LATCH]] ], [ [[TMP132]], [[MIDDLE_BLOCK]] ], [ [[TMP169]], [[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-NEXT: [[MERGE_LCSSA:%.*]] = phi i64 [ [[MERGE]], [[LOOP_LATCH1]] ], [ [[TMP132]], [[MIDDLE_BLOCK]] ], [ [[TMP169]], [[VEC_EPILOG_MIDDLE_BLOCK]] ]
; CHECK-NEXT: ret i64 [[MERGE_LCSSA]]
;
entry:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll b/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll
index 2c740670be578..4f9f683195817 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll
@@ -454,7 +454,7 @@ define void @multi_exit(ptr %dst, ptr %src.1, ptr %src.2, i64 %A, i64 %B) #0 {
; CHECK-NEXT: [[TMP1:%.*]] = freeze i64 [[TMP0]]
; CHECK-NEXT: [[UMIN10:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP1]], i64 [[A]])
; CHECK-NEXT: [[TMP2:%.*]] = add nuw i64 [[UMIN10]], 1
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ule i64 [[TMP2]], 24
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ule i64 [[TMP2]], 32
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
; CHECK: [[VECTOR_SCEVCHECK]]:
; CHECK-NEXT: [[TMP3:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[B]], i64 1)
@@ -488,31 +488,31 @@ define void @multi_exit(ptr %dst, ptr %src.1, ptr %src.2, i64 %A, i64 %B) #0 {
; CHECK-NEXT: [[CONFLICT_RDX:%.*]] = or i1 [[FOUND_CONFLICT]], [[FOUND_CONFLICT8]]
; CHECK-NEXT: br i1 [[CONFLICT_RDX]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i64 [[TMP2]], 3
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i64 [[TMP2]], 31
; CHECK-NEXT: [[TMP19:%.*]] = icmp eq i64 [[N_MOD_VF]], 0
-; CHECK-NEXT: [[TMP20:%.*]] = select i1 [[TMP19]], i64 4, i64 [[N_MOD_VF]]
+; CHECK-NEXT: [[TMP20:%.*]] = select i1 [[TMP19]], i64 32, i64 [[N_MOD_VF]]
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP2]], [[TMP20]]
; CHECK-NEXT: [[TMP21:%.*]] = trunc i64 [[N_VEC]] to i32
; CHECK-NEXT: [[TMP22:%.*]] = load i64, ptr [[SRC_2]], align 8, !alias.scope [[META6:![0-9]+]]
; CHECK-NEXT: [[TMP23:%.*]] = icmp ne i64 [[TMP22]], 0
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <2 x i1> poison, i1 [[TMP23]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <2 x i1> [[BROADCAST_SPLATINSERT]], <2 x i1> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <16 x i1> poison, i1 [[TMP23]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <16 x i1> [[BROADCAST_SPLATINSERT]], <16 x i1> poison, <16 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP24:%.*]] = trunc i64 [[INDEX]] to i32
; CHECK-NEXT: [[TMP25:%.*]] = getelementptr inbounds i64, ptr [[SRC_1]], i32 [[TMP24]]
-; CHECK-NEXT: [[TMP26:%.*]] = getelementptr inbounds i64, ptr [[TMP25]], i64 2
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x i64>, ptr [[TMP26]], align 8, !alias.scope [[META9:![0-9]+]]
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP26:%.*]] = getelementptr inbounds i64, ptr [[TMP25]], i64 16
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i64>, ptr [[TMP26]], align 8, !alias.scope [[META9:![0-9]+]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; CHECK-NEXT: [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: [[TMP27:%.*]] = icmp eq <2 x i64> [[WIDE_LOAD]], zeroinitializer
-; CHECK-NEXT: [[TMP28:%.*]] = and <2 x i1> [[BROADCAST_SPLAT]], [[TMP27]]
-; CHECK-NEXT: [[TMP29:%.*]] = zext <2 x i1> [[TMP28]] to <2 x i8>
-; CHECK-NEXT: [[TMP30:%.*]] = extractelement <2 x i8> [[TMP29]], i64 1
-; CHECK-NEXT: store i8 [[TMP30]], ptr [[DST]], align 1, !alias.scope [[META12:![0-9]+]], !noalias [[META14:![0-9]+]]
+; CHECK-NEXT: [[TMP28:%.*]] = icmp eq <16 x i64> [[WIDE_LOAD]], zeroinitializer
+; CHECK-NEXT: [[TMP29:%.*]] = and <16 x i1> [[BROADCAST_SPLAT]], [[TMP28]]
+; CHECK-NEXT: [[TMP30:%.*]] = zext <16 x i1> [[TMP29]] to <16 x i8>
+; CHECK-NEXT: [[TMP32:%.*]] = extractelement <16 x i8> [[TMP30]], i64 15
+; CHECK-NEXT: store i8 [[TMP32]], ptr [[DST]], align 1, !alias.scope [[META12:![0-9]+]], !noalias [[META14:![0-9]+]]
; CHECK-NEXT: br label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
diff --git a/llvm/test/Transforms/LoopVectorize/X86/gcc-examples.ll b/llvm/test/Transforms/LoopVectorize/X86/gcc-examples.ll
index 05e855bc338d2..c9f1afcc61e4c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/gcc-examples.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/gcc-examples.ll
@@ -1,3 +1,4 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
; RUN: opt < %s -passes=loop-vectorize -mtriple=x86_64-apple-macosx10.8.0 -mcpu=corei7 -S | FileCheck %s
; RUN: opt < %s -passes=loop-vectorize -mtriple=x86_64-apple-macosx10.8.0 -mcpu=corei7 -force-vector-interleave=0 -S | FileCheck %s -check-prefix=UNROLL
@@ -9,21 +10,66 @@ target triple = "x86_64-apple-macosx10.8.0"
@a = common global [2048 x i32] zeroinitializer, align 16
; Select VF = 8;
-;CHECK-LABEL: @example1(
-;CHECK: load <4 x i32>
-;CHECK: add nsw <4 x i32>
-;CHECK: store <4 x i32>
-;CHECK: ret void
-;UNROLL-LABEL: @example1(
-;UNROLL: load <4 x i32>
-;UNROLL: load <4 x i32>
-;UNROLL: add nsw <4 x i32>
-;UNROLL: add nsw <4 x i32>
-;UNROLL: store <4 x i32>
-;UNROLL: store <4 x i32>
-;UNROLL: ret void
define void @example1() {
+; CHECK-LABEL: define void @example1(
+; CHECK-SAME: ) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds [2048 x i32], ptr @b, i64 0, i64 [[INDEX]]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 4
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds [2048 x i32], ptr @c, i64 0, i64 [[INDEX]]
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP3]], i64 4
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = add nsw <4 x i32> [[WIDE_LOAD2]], [[WIDE_LOAD]]
+; CHECK-NEXT: [[TMP6:%.*]] = add nsw <4 x i32> [[WIDE_LOAD3]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds [2048 x i32], ptr @a, i64 0, i64 [[INDEX]]
+; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 4
+; CHECK-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP7]], align 4
+; CHECK-NEXT: store <4 x i32> [[TMP6]], ptr [[TMP8]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256
+; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[BB10:.*]]
+; CHECK: [[BB10]]:
+; CHECK-NEXT: ret void
+;
+; UNROLL-LABEL: define void @example1(
+; UNROLL-SAME: ) #[[ATTR0:[0-9]+]] {
+; UNROLL-NEXT: br label %[[VECTOR_PH:.*]]
+; UNROLL: [[VECTOR_PH]]:
+; UNROLL-NEXT: br label %[[VECTOR_BODY:.*]]
+; UNROLL: [[VECTOR_BODY]]:
+; UNROLL-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNROLL-NEXT: [[TMP1:%.*]] = getelementptr inbounds [2048 x i32], ptr @b, i64 0, i64 [[INDEX]]
+; UNROLL-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 4
+; UNROLL-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; UNROLL-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; UNROLL-NEXT: [[TMP3:%.*]] = getelementptr inbounds [2048 x i32], ptr @c, i64 0, i64 [[INDEX]]
+; UNROLL-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP3]], i64 4
+; UNROLL-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; UNROLL-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; UNROLL-NEXT: [[TMP5:%.*]] = add nsw <4 x i32> [[WIDE_LOAD2]], [[WIDE_LOAD]]
+; UNROLL-NEXT: [[TMP6:%.*]] = add nsw <4 x i32> [[WIDE_LOAD3]], [[WIDE_LOAD1]]
+; UNROLL-NEXT: [[TMP7:%.*]] = getelementptr inbounds [2048 x i32], ptr @a, i64 0, i64 [[INDEX]]
+; UNROLL-NEXT: [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 4
+; UNROLL-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP7]], align 4
+; UNROLL-NEXT: store <4 x i32> [[TMP6]], ptr [[TMP8]], align 4
+; UNROLL-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; UNROLL-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256
+; UNROLL-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; UNROLL: [[MIDDLE_BLOCK]]:
+; UNROLL-NEXT: br label %[[BB10:.*]]
+; UNROLL: [[BB10]]:
+; UNROLL-NEXT: ret void
+;
br label %1
; <label>:
@@ -45,18 +91,57 @@ define void @example1() {
}
; Select VF=4 because sext <8 x i1> to <8 x i32> is expensive.
-;CHECK-LABEL: @example10b(
-;CHECK: load <4 x i16>
-;CHECK: sext <4 x i16>
-;CHECK: store <4 x i32>
-;CHECK: ret void
-;UNROLL-LABEL: @example10b(
-;UNROLL: load <4 x i16>
-;UNROLL: load <4 x i16>
-;UNROLL: store <4 x i32>
-;UNROLL: store <4 x i32>
-;UNROLL: ret void
define void @example10b(ptr noalias nocapture %sa, ptr noalias nocapture %sb, ptr noalias nocapture %sc, ptr noalias nocapture %ia, ptr noalias nocapture %ib, ptr noalias nocapture %ic) {
+; CHECK-LABEL: define void @example10b(
+; CHECK-SAME: ptr noalias captures(none) [[SA:%.*]], ptr noalias captures(none) [[SB:%.*]], ptr noalias captures(none) [[SC:%.*]], ptr noalias captures(none) [[IA:%.*]], ptr noalias captures(none) [[IB:%.*]], ptr noalias captures(none) [[IC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i16, ptr [[SB]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i16, ptr [[TMP1]], i64 8
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i16>, ptr [[TMP1]], align 2
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <8 x i16>, ptr [[TMP2]], align 2
+; CHECK-NEXT: [[TMP3:%.*]] = sext <8 x i16> [[WIDE_LOAD]] to <8 x i32>
+; CHECK-NEXT: [[TMP4:%.*]] = sext <8 x i16> [[WIDE_LOAD1]] to <8 x i32>
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[IA]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[TMP5]], i64 8
+; CHECK-NEXT: store <8 x i32> [[TMP3]], ptr [[TMP5]], align 4
+; CHECK-NEXT: store <8 x i32> [[TMP4]], ptr [[TMP6]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[BB8:.*]]
+; CHECK: [[BB8]]:
+; CHECK-NEXT: ret void
+;
+; UNROLL-LABEL: define void @example10b(
+; UNROLL-SAME: ptr noalias captures(none) [[SA:%.*]], ptr noalias captures(none) [[SB:%.*]], ptr noalias captures(none) [[SC:%.*]], ptr noalias captures(none) [[IA:%.*]], ptr noalias captures(none) [[IB:%.*]], ptr noalias captures(none) [[IC:%.*]]) #[[ATTR0]] {
+; UNROLL-NEXT: br label %[[VECTOR_PH:.*]]
+; UNROLL: [[VECTOR_PH]]:
+; UNROLL-NEXT: br label %[[VECTOR_BODY:.*]]
+; UNROLL: [[VECTOR_BODY]]:
+; UNROLL-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNROLL-NEXT: [[TMP1:%.*]] = getelementptr inbounds i16, ptr [[SB]], i64 [[INDEX]]
+; UNROLL-NEXT: [[TMP2:%.*]] = getelementptr inbounds i16, ptr [[TMP1]], i64 8
+; UNROLL-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i16>, ptr [[TMP1]], align 2
+; UNROLL-NEXT: [[WIDE_LOAD1:%.*]] = load <8 x i16>, ptr [[TMP2]], align 2
+; UNROLL-NEXT: [[TMP3:%.*]] = sext <8 x i16> [[WIDE_LOAD]] to <8 x i32>
+; UNROLL-NEXT: [[TMP4:%.*]] = sext <8 x i16> [[WIDE_LOAD1]] to <8 x i32>
+; UNROLL-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[IA]], i64 [[INDEX]]
+; UNROLL-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[TMP5]], i64 8
+; UNROLL-NEXT: store <8 x i32> [[TMP3]], ptr [[TMP5]], align 4
+; UNROLL-NEXT: store <8 x i32> [[TMP4]], ptr [[TMP6]], align 4
+; UNROLL-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; UNROLL-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; UNROLL-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; UNROLL: [[MIDDLE_BLOCK]]:
+; UNROLL-NEXT: br label %[[BB8:.*]]
+; UNROLL: [[BB8]]:
+; UNROLL-NEXT: ret void
+;
br label %1
; <label>:
@@ -75,3 +160,14 @@ define void @example10b(ptr noalias nocapture %sa, ptr noalias nocapture %sb, pt
ret void
}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META1]], [[META2]]}
+;.
+; UNROLL: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; UNROLL: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; UNROLL: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; UNROLL: [[LOOP3]] = distinct !{[[LOOP3]], [[META1]], [[META2]]}
+;.
diff --git a/llvm/test/Transforms/LoopVectorize/X86/induction-costs.ll b/llvm/test/Transforms/LoopVectorize/X86/induction-costs.ll
index 9029f4d833178..700c2b610b480 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/induction-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/induction-costs.ll
@@ -118,23 +118,23 @@ define void @multiple_truncated_ivs_with_wide_uses(i1 %c, ptr %A, ptr %B) {
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i16> [ <i16 0, i16 1, i16 2, i16 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND3:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT6:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[STEP_ADD:%.*]] = add <4 x i16> [[VEC_IND]], splat (i16 4)
-; CHECK-NEXT: [[STEP_ADD4:%.*]] = add <4 x i32> [[VEC_IND3]], splat (i32 4)
-; CHECK-NEXT: [[TMP1:%.*]] = select i1 [[C]], <4 x i16> [[VEC_IND]], <4 x i16> splat (i16 10)
-; CHECK-NEXT: [[TMP2:%.*]] = select i1 [[C]], <4 x i16> [[STEP_ADD]], <4 x i16> splat (i16 10)
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <8 x i16> [ <i16 0, i16 1, i16 2, i16 3, i16 4, i16 5, i16 6, i16 7>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND2:%.*]] = phi <8 x i32> [ <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT4:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[STEP_ADD:%.*]] = add <8 x i16> [[VEC_IND]], splat (i16 8)
+; CHECK-NEXT: [[STEP_ADD3:%.*]] = add <8 x i32> [[VEC_IND2]], splat (i32 8)
+; CHECK-NEXT: [[TMP0:%.*]] = select i1 [[C]], <8 x i16> [[VEC_IND]], <8 x i16> splat (i16 10)
+; CHECK-NEXT: [[TMP1:%.*]] = select i1 [[C]], <8 x i16> [[STEP_ADD]], <8 x i16> splat (i16 10)
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr i16, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i16, ptr [[TMP4]], i64 4
-; CHECK-NEXT: store <4 x i16> [[TMP1]], ptr [[TMP4]], align 2, !alias.scope [[META6:![0-9]+]], !noalias [[META9:![0-9]+]]
-; CHECK-NEXT: store <4 x i16> [[TMP2]], ptr [[TMP3]], align 2, !alias.scope [[META6]], !noalias [[META9]]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i16, ptr [[TMP4]], i64 8
+; CHECK-NEXT: store <8 x i16> [[TMP0]], ptr [[TMP4]], align 2, !alias.scope [[META6:![0-9]+]], !noalias [[META9:![0-9]+]]
+; CHECK-NEXT: store <8 x i16> [[TMP1]], ptr [[TMP3]], align 2, !alias.scope [[META6]], !noalias [[META9]]
; CHECK-NEXT: [[TMP8:%.*]] = getelementptr i32, ptr [[B]], i64 [[INDEX]]
-; CHECK-NEXT: [[TMP5:%.*]] = getelementptr i32, ptr [[TMP8]], i64 4
-; CHECK-NEXT: store <4 x i32> [[VEC_IND3]], ptr [[TMP8]], align 4, !alias.scope [[META9]]
-; CHECK-NEXT: store <4 x i32> [[STEP_ADD4]], ptr [[TMP5]], align 4, !alias.scope [[META9]]
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i16> [[STEP_ADD]], splat (i16 4)
-; CHECK-NEXT: [[VEC_IND_NEXT6]] = add <4 x i32> [[STEP_ADD4]], splat (i32 4)
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr i32, ptr [[TMP8]], i64 8
+; CHECK-NEXT: store <8 x i32> [[VEC_IND2]], ptr [[TMP8]], align 4, !alias.scope [[META9]]
+; CHECK-NEXT: store <8 x i32> [[STEP_ADD3]], ptr [[TMP5]], align 4, !alias.scope [[META9]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add <8 x i16> [[STEP_ADD]], splat (i16 8)
+; CHECK-NEXT: [[VEC_IND_NEXT4]] = add <8 x i32> [[STEP_ADD3]], splat (i32 8)
; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], 64
; CHECK-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
@@ -515,7 +515,7 @@ define i32 @test_scalar_predicated_cost(i64 %x, i64 %y, ptr %A) #0 {
; CHECK-NEXT: [[INDEX_NEXT11]] = add nuw i64 [[INDEX4]], 4
; CHECK-NEXT: [[VEC_IND_NEXT6]] = add <4 x i64> [[VEC_IND5]], splat (i64 4)
; CHECK-NEXT: [[TMP30:%.*]] = icmp eq i64 [[INDEX_NEXT11]], 100
-; CHECK-NEXT: br i1 [[TMP30]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP25:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP30]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP26:![0-9]+]]
; CHECK: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; CHECK-NEXT: br i1 false, label %[[EXIT]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
@@ -534,7 +534,7 @@ define i32 @test_scalar_predicated_cost(i64 %x, i64 %y, ptr %A) #0 {
; CHECK: [[LOOP_LATCH]]:
; CHECK-NEXT: [[IV_NEXT]] = add i64 [[IV]], 1
; CHECK-NEXT: [[EC:%.*]] = icmp eq i64 [[IV]], 100
-; CHECK-NEXT: br i1 [[EC]], label %[[EXIT]], label %[[LOOP_HEADER1]], !llvm.loop [[LOOP26:![0-9]+]]
+; CHECK-NEXT: br i1 [[EC]], label %[[EXIT]], label %[[LOOP_HEADER1]], !llvm.loop [[LOOP27:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret i32 0
;
@@ -614,7 +614,7 @@ define void @wide_iv_trunc(ptr %dst, i64 %N) {
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT: br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP27:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP28:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: br label %[[EXIT_LOOPEXIT:.*]]
; CHECK: [[EXIT_LOOPEXIT]]:
@@ -709,11 +709,11 @@ define void @wombat(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; CHECK-NEXT: [[VEC_IND_NEXT]] = add <8 x i32> [[STEP_ADD]], [[TMP2]]
; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], 48
-; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP28:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP29:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: br i1 false, label %[[EXIT:.*]], label %[[VEC_EPILOG_ENTRY:.*]]
; CHECK: [[VEC_EPILOG_ENTRY]]:
-; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF29:![0-9]+]]
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF30:![0-9]+]]
; CHECK: [[VEC_EPILOG_PH]]:
; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ 48, %[[VEC_EPILOG_ENTRY]] ], [ 0, %[[VECTOR_MAIN_LOOP_ENTRY]] ]
; CHECK-NEXT: [[BC_RESUME_VAL3:%.*]] = phi i32 [ [[IND_END]], %[[VEC_EPILOG_ENTRY]] ], [ [[MUL]], %[[VECTOR_MAIN_LOOP_ENTRY]] ]
@@ -739,7 +739,7 @@ define void @wombat(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX11]], 4
; CHECK-NEXT: [[VEC_IND_NEXT14]] = add <4 x i32> [[VEC_IND12]], [[BROADCAST_SPLAT10]]
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT13]], 60
-; CHECK-NEXT: br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP30:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP31:![0-9]+]]
; CHECK: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; CHECK-NEXT: br i1 false, label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
; CHECK: [[VEC_EPILOG_SCALAR_PH]]:
@@ -756,7 +756,7 @@ define void @wombat(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[ADD]] = add i64 [[PHI]], 1
; CHECK-NEXT: [[ICMP:%.*]] = icmp ugt i64 [[PHI]], 65
; CHECK-NEXT: [[TRUNC]] = trunc i64 [[MUL3]] to i32
-; CHECK-NEXT: br i1 [[ICMP]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP31:![0-9]+]]
+; CHECK-NEXT: br i1 [[ICMP]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP32:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -815,11 +815,11 @@ define void @wombat2(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; CHECK-NEXT: [[VEC_IND_NEXT]] = add <8 x i32> [[STEP_ADD]], [[TMP2]]
; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], 48
-; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: br i1 false, label %[[EXIT:.*]], label %[[VEC_EPILOG_ENTRY:.*]]
; CHECK: [[VEC_EPILOG_ENTRY]]:
-; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF29]]
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF30]]
; CHECK: [[VEC_EPILOG_PH]]:
; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ 48, %[[VEC_EPILOG_ENTRY]] ], [ 0, %[[VECTOR_MAIN_LOOP_ENTRY]] ]
; CHECK-NEXT: [[BC_RESUME_VAL3:%.*]] = phi i32 [ [[IND_END]], %[[VEC_EPILOG_ENTRY]] ], [ [[MUL]], %[[VECTOR_MAIN_LOOP_ENTRY]] ]
@@ -845,7 +845,7 @@ define void @wombat2(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX11]], 4
; CHECK-NEXT: [[VEC_IND_NEXT14]] = add <4 x i32> [[VEC_IND12]], [[BROADCAST_SPLAT10]]
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT13]], 60
-; CHECK-NEXT: br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP34:![0-9]+]]
; CHECK: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; CHECK-NEXT: br i1 false, label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
; CHECK: [[VEC_EPILOG_SCALAR_PH]]:
@@ -863,7 +863,7 @@ define void @wombat2(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[ICMP:%.*]] = icmp ugt i64 [[PHI]], 65
; CHECK-NEXT: [[TRUNC_0:%.*]] = trunc i64 [[MUL3]] to i60
; CHECK-NEXT: [[TRUNC_1]] = trunc i60 [[TRUNC_0]] to i32
-; CHECK-NEXT: br i1 [[ICMP]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP34:![0-9]+]]
+; CHECK-NEXT: br i1 [[ICMP]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP35:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -924,11 +924,11 @@ define void @with_dead_use(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; CHECK-NEXT: [[VEC_IND_NEXT]] = add <8 x i32> [[STEP_ADD]], [[TMP2]]
; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], 48
-; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP35:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: br i1 false, label %[[EXIT:.*]], label %[[VEC_EPILOG_ENTRY:.*]]
; CHECK: [[VEC_EPILOG_ENTRY]]:
-; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF29]]
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF30]]
; CHECK: [[VEC_EPILOG_PH]]:
; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ 48, %[[VEC_EPILOG_ENTRY]] ], [ 0, %[[VECTOR_MAIN_LOOP_ENTRY]] ]
; CHECK-NEXT: [[BC_RESUME_VAL3:%.*]] = phi i32 [ [[IND_END]], %[[VEC_EPILOG_ENTRY]] ], [ [[MUL]], %[[VECTOR_MAIN_LOOP_ENTRY]] ]
@@ -954,7 +954,7 @@ define void @with_dead_use(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX11]], 4
; CHECK-NEXT: [[VEC_IND_NEXT14]] = add <4 x i32> [[VEC_IND12]], [[BROADCAST_SPLAT10]]
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT13]], 60
-; CHECK-NEXT: br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP37:![0-9]+]]
; CHECK: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; CHECK-NEXT: br i1 false, label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
; CHECK: [[VEC_EPILOG_SCALAR_PH]]:
@@ -972,7 +972,7 @@ define void @with_dead_use(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[ICMP:%.*]] = icmp ugt i64 [[PHI]], 65
; CHECK-NEXT: [[TRUNC]] = trunc i64 [[MUL3]] to i32
; CHECK-NEXT: [[DEAD_AND:%.*]] = and i32 [[TRUNC]], 123
-; CHECK-NEXT: br i1 [[ICMP]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP37:![0-9]+]]
+; CHECK-NEXT: br i1 [[ICMP]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP38:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
diff --git a/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll b/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll
index 310d894c159ad..806a81f721212 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll
@@ -813,42 +813,15 @@ define void @foo3(ptr nocapture %A, ptr nocapture readonly %B, ptr nocapture rea
; AVX2: [[VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 4
-; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 8
-; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 12
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
-; AVX2-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4, !alias.scope [[META12]]
-; AVX2-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META12]]
-; AVX2-NEXT: [[WIDE_LOAD8:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META12]]
-; AVX2-NEXT: [[TMP4:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD]], splat (i32 100)
-; AVX2-NEXT: [[TMP5:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD6]], splat (i32 100)
-; AVX2-NEXT: [[TMP6:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD7]], splat (i32 100)
-; AVX2-NEXT: [[TMP7:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD8]], splat (i32 100)
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
+; AVX2-NEXT: [[TMP1:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD]], splat (i32 100)
; AVX2-NEXT: [[TMP8:%.*]] = getelementptr double, ptr [[B]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP9:%.*]] = getelementptr double, ptr [[TMP8]], i64 4
-; AVX2-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[TMP8]], i64 8
-; AVX2-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP8]], i64 12
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP8]], <4 x i1> [[TMP4]], <4 x double> poison), !alias.scope [[META15:![0-9]+]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD9:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP9]], <4 x i1> [[TMP5]], <4 x double> poison), !alias.scope [[META15]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD10:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP10]], <4 x i1> [[TMP6]], <4 x double> poison), !alias.scope [[META15]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP11]], <4 x i1> [[TMP7]], <4 x double> poison), !alias.scope [[META15]]
-; AVX2-NEXT: [[TMP12:%.*]] = sitofp <4 x i32> [[WIDE_LOAD]] to <4 x double>
-; AVX2-NEXT: [[TMP13:%.*]] = sitofp <4 x i32> [[WIDE_LOAD6]] to <4 x double>
-; AVX2-NEXT: [[TMP14:%.*]] = sitofp <4 x i32> [[WIDE_LOAD7]] to <4 x double>
-; AVX2-NEXT: [[TMP15:%.*]] = sitofp <4 x i32> [[WIDE_LOAD8]] to <4 x double>
-; AVX2-NEXT: [[TMP16:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD]], [[TMP12]]
-; AVX2-NEXT: [[TMP17:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD9]], [[TMP13]]
-; AVX2-NEXT: [[TMP18:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD10]], [[TMP14]]
-; AVX2-NEXT: [[TMP19:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD11]], [[TMP15]]
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP8]], <8 x i1> [[TMP1]], <8 x double> poison), !alias.scope [[META15:![0-9]+]]
+; AVX2-NEXT: [[TMP3:%.*]] = sitofp <8 x i32> [[WIDE_LOAD]] to <8 x double>
+; AVX2-NEXT: [[TMP4:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], [[TMP3]]
; AVX2-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[A]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP20]], i64 4
-; AVX2-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP20]], i64 8
-; AVX2-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP20]], i64 12
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP16]], ptr align 8 [[TMP20]], <4 x i1> [[TMP4]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP17]], ptr align 8 [[TMP21]], <4 x i1> [[TMP5]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP18]], ptr align 8 [[TMP22]], <4 x i1> [[TMP6]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP19]], ptr align 8 [[TMP23]], <4 x i1> [[TMP7]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP4]], ptr align 8 [[TMP20]], <8 x i1> [[TMP1]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; AVX2-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 10000
; AVX2-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
@@ -878,65 +851,65 @@ define void @foo3(ptr nocapture %A, ptr nocapture readonly %B, ptr nocapture rea
; AVX512: [[VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 8
; AVX512-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 16
-; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 24
-; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
-; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <8 x i32>, ptr [[TMP1]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <8 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[TMP4:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD]], splat (i32 100)
-; AVX512-NEXT: [[TMP5:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD6]], splat (i32 100)
-; AVX512-NEXT: [[TMP6:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD7]], splat (i32 100)
-; AVX512-NEXT: [[TMP7:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD8]], splat (i32 100)
+; AVX512-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 32
+; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 48
+; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
+; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <16 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <16 x i32>, ptr [[TMP9]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <16 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[TMP4:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD]], splat (i32 100)
+; AVX512-NEXT: [[TMP5:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD6]], splat (i32 100)
+; AVX512-NEXT: [[TMP6:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD7]], splat (i32 100)
+; AVX512-NEXT: [[TMP7:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD8]], splat (i32 100)
; AVX512-NEXT: [[TMP8:%.*]] = getelementptr double, ptr [[B]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP9:%.*]] = getelementptr double, ptr [[TMP8]], i64 8
; AVX512-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[TMP8]], i64 16
-; AVX512-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP8]], i64 24
-; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP8]], <8 x i1> [[TMP4]], <8 x double> poison), !alias.scope [[META15:![0-9]+]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD9:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP9]], <8 x i1> [[TMP5]], <8 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD10:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP10]], <8 x i1> [[TMP6]], <8 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP11]], <8 x i1> [[TMP7]], <8 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[TMP12:%.*]] = sitofp <8 x i32> [[WIDE_LOAD]] to <8 x double>
-; AVX512-NEXT: [[TMP13:%.*]] = sitofp <8 x i32> [[WIDE_LOAD6]] to <8 x double>
-; AVX512-NEXT: [[TMP14:%.*]] = sitofp <8 x i32> [[WIDE_LOAD7]] to <8 x double>
-; AVX512-NEXT: [[TMP15:%.*]] = sitofp <8 x i32> [[WIDE_LOAD8]] to <8 x double>
-; AVX512-NEXT: [[TMP16:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], [[TMP12]]
-; AVX512-NEXT: [[TMP17:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD9]], [[TMP13]]
-; AVX512-NEXT: [[TMP18:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD10]], [[TMP14]]
-; AVX512-NEXT: [[TMP19:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD11]], [[TMP15]]
+; AVX512-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP8]], i64 32
+; AVX512-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP8]], i64 48
+; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP8]], <16 x i1> [[TMP4]], <16 x double> poison), !alias.scope [[META15:![0-9]+]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD9:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP10]], <16 x i1> [[TMP5]], <16 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD10:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP21]], <16 x i1> [[TMP6]], <16 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP11]], <16 x i1> [[TMP7]], <16 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[TMP12:%.*]] = sitofp <16 x i32> [[WIDE_LOAD]] to <16 x double>
+; AVX512-NEXT: [[TMP13:%.*]] = sitofp <16 x i32> [[WIDE_LOAD6]] to <16 x double>
+; AVX512-NEXT: [[TMP14:%.*]] = sitofp <16 x i32> [[WIDE_LOAD7]] to <16 x double>
+; AVX512-NEXT: [[TMP15:%.*]] = sitofp <16 x i32> [[WIDE_LOAD8]] to <16 x double>
+; AVX512-NEXT: [[TMP16:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD]], [[TMP12]]
+; AVX512-NEXT: [[TMP17:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD9]], [[TMP13]]
+; AVX512-NEXT: [[TMP18:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD10]], [[TMP14]]
+; AVX512-NEXT: [[TMP19:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD11]], [[TMP15]]
; AVX512-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[A]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP20]], i64 8
; AVX512-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP20]], i64 16
-; AVX512-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP20]], i64 24
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP16]], ptr align 8 [[TMP20]], <8 x i1> [[TMP4]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP17]], ptr align 8 [[TMP21]], <8 x i1> [[TMP5]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP18]], ptr align 8 [[TMP22]], <8 x i1> [[TMP6]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP19]], ptr align 8 [[TMP23]], <8 x i1> [[TMP7]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX512-NEXT: [[TMP32:%.*]] = getelementptr double, ptr [[TMP20]], i64 32
+; AVX512-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP20]], i64 48
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP16]], ptr align 8 [[TMP20]], <16 x i1> [[TMP4]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP17]], ptr align 8 [[TMP22]], <16 x i1> [[TMP5]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP18]], ptr align 8 [[TMP32]], <16 x i1> [[TMP6]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP19]], ptr align 8 [[TMP23]], <16 x i1> [[TMP7]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
; AVX512-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 9984
; AVX512-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: br i1 false, [[FOR_END:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX512: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX512-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF21:![0-9]+]]
+; AVX512-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
; AVX512: [[VEC_EPILOG_PH]]:
; AVX512-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ 9984, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX512-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX512: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX12:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT15:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP25:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[INDEX12]]
-; AVX512-NEXT: [[WIDE_LOAD13:%.*]] = load <8 x i32>, ptr [[TMP25]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[TMP26:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD13]], splat (i32 100)
+; AVX512-NEXT: [[WIDE_LOAD13:%.*]] = load <16 x i32>, ptr [[TMP25]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[TMP26:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD13]], splat (i32 100)
; AVX512-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[B]], i64 [[INDEX12]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD14:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP27]], <8 x i1> [[TMP26]], <8 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[TMP28:%.*]] = sitofp <8 x i32> [[WIDE_LOAD13]] to <8 x double>
-; AVX512-NEXT: [[TMP29:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD14]], [[TMP28]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD14:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP27]], <16 x i1> [[TMP26]], <16 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[TMP28:%.*]] = sitofp <16 x i32> [[WIDE_LOAD13]] to <16 x double>
+; AVX512-NEXT: [[TMP29:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD14]], [[TMP28]]
; AVX512-NEXT: [[TMP30:%.*]] = getelementptr double, ptr [[A]], i64 [[INDEX12]]
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP29]], ptr align 8 [[TMP30]], <8 x i1> [[TMP26]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: [[INDEX_NEXT15]] = add nuw i64 [[INDEX12]], 8
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP29]], ptr align 8 [[TMP30]], <16 x i1> [[TMP26]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: [[INDEX_NEXT15]] = add nuw i64 [[INDEX12]], 16
; AVX512-NEXT: [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT15]], 10000
-; AVX512-NEXT: br i1 [[TMP31]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP22:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP31]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP21:![0-9]+]]
; AVX512: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX512-NEXT: br i1 true, [[FOR_END]], label %[[VEC_EPILOG_SCALAR_PH]]
; AVX512: [[VEC_EPILOG_SCALAR_PH]]:
@@ -1027,21 +1000,21 @@ define void @foo4(ptr nocapture %A, ptr nocapture readonly %B, ptr nocapture rea
; AVX512-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX512: [[VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; AVX512-NEXT: [[VEC_IND:%.*]] = phi <8 x i64> [ <i64 0, i64 16, i64 32, i64 48, i64 64, i64 80, i64 96, i64 112>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; AVX512-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], <8 x i64> [[VEC_IND]]
-; AVX512-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <8 x i32> @llvm.masked.gather.v8i32.v8p0(<8 x ptr> align 4 [[WIDE_GEP]], <8 x i1> splat (i1 true), <8 x i32> poison), !alias.scope [[META24:![0-9]+]]
-; AVX512-NEXT: [[TMP0:%.*]] = icmp slt <8 x i32> [[WIDE_MASKED_GATHER]], splat (i32 100)
-; AVX512-NEXT: [[TMP1:%.*]] = shl nuw nsw <8 x i64> [[VEC_IND]], splat (i64 1)
-; AVX512-NEXT: [[WIDE_GEP6:%.*]] = getelementptr inbounds double, ptr [[B]], <8 x i64> [[TMP1]]
-; AVX512-NEXT: [[WIDE_MASKED_GATHER7:%.*]] = call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 8 [[WIDE_GEP6]], <8 x i1> [[TMP0]], <8 x double> poison), !alias.scope [[META27:![0-9]+]]
-; AVX512-NEXT: [[TMP2:%.*]] = sitofp <8 x i32> [[WIDE_MASKED_GATHER]] to <8 x double>
-; AVX512-NEXT: [[TMP3:%.*]] = fadd <8 x double> [[WIDE_MASKED_GATHER7]], [[TMP2]]
-; AVX512-NEXT: [[WIDE_GEP8:%.*]] = getelementptr inbounds double, ptr [[A]], <8 x i64> [[VEC_IND]]
-; AVX512-NEXT: call void @llvm.masked.scatter.v8f64.v8p0(<8 x double> [[TMP3]], <8 x ptr> align 8 [[WIDE_GEP8]], <8 x i1> [[TMP0]]), !alias.scope [[META29:![0-9]+]], !noalias [[META31:![0-9]+]]
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
-; AVX512-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <8 x i64> [[VEC_IND]], splat (i64 128)
+; AVX512-NEXT: [[VEC_IND:%.*]] = phi <16 x i64> [ <i64 0, i64 16, i64 32, i64 48, i64 64, i64 80, i64 96, i64 112, i64 128, i64 144, i64 160, i64 176, i64 192, i64 208, i64 224, i64 240>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; AVX512-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], <16 x i64> [[VEC_IND]]
+; AVX512-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <16 x i32> @llvm.masked.gather.v16i32.v16p0(<16 x ptr> align 4 [[WIDE_GEP]], <16 x i1> splat (i1 true), <16 x i32> poison), !alias.scope [[META23:![0-9]+]]
+; AVX512-NEXT: [[TMP0:%.*]] = icmp slt <16 x i32> [[WIDE_MASKED_GATHER]], splat (i32 100)
+; AVX512-NEXT: [[TMP1:%.*]] = shl nuw nsw <16 x i64> [[VEC_IND]], splat (i64 1)
+; AVX512-NEXT: [[WIDE_GEP6:%.*]] = getelementptr inbounds double, ptr [[B]], <16 x i64> [[TMP1]]
+; AVX512-NEXT: [[WIDE_MASKED_GATHER7:%.*]] = call <16 x double> @llvm.masked.gather.v16f64.v16p0(<16 x ptr> align 8 [[WIDE_GEP6]], <16 x i1> [[TMP0]], <16 x double> poison), !alias.scope [[META26:![0-9]+]]
+; AVX512-NEXT: [[TMP2:%.*]] = sitofp <16 x i32> [[WIDE_MASKED_GATHER]] to <16 x double>
+; AVX512-NEXT: [[TMP3:%.*]] = fadd <16 x double> [[WIDE_MASKED_GATHER7]], [[TMP2]]
+; AVX512-NEXT: [[WIDE_GEP8:%.*]] = getelementptr inbounds double, ptr [[A]], <16 x i64> [[VEC_IND]]
+; AVX512-NEXT: call void @llvm.masked.scatter.v16f64.v16p0(<16 x double> [[TMP3]], <16 x ptr> align 8 [[WIDE_GEP8]], <16 x i1> [[TMP0]]), !alias.scope [[META28:![0-9]+]], !noalias [[META30:![0-9]+]]
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX512-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <16 x i64> [[VEC_IND]], splat (i64 256)
; AVX512-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], 624
-; AVX512-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP31:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: br label %[[SCALAR_PH]]
; AVX512: [[SCALAR_PH]]:
@@ -1111,49 +1084,19 @@ define void @foo6(ptr nocapture readonly %in, ptr nocapture %out, i32 %size, ptr
; AVX1-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP0:%.*]] = sub i64 4095, [[INDEX]]
; AVX1-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[TMP0]]
-; AVX1-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -3
; AVX1-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -7
-; AVX1-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -11
-; AVX1-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -15
-; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META18:![0-9]+]]
-; AVX1-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META18]]
-; AVX1-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META18]]
-; AVX1-NEXT: [[WIDE_LOAD8:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META18]]
-; AVX1-NEXT: [[REVERSE:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[REVERSE9:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD6]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[REVERSE10:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD7]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[REVERSE11:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD8]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[TMP6:%.*]] = icmp sgt <4 x i32> [[REVERSE]], zeroinitializer
-; AVX1-NEXT: [[TMP7:%.*]] = icmp sgt <4 x i32> [[REVERSE9]], zeroinitializer
-; AVX1-NEXT: [[TMP8:%.*]] = icmp sgt <4 x i32> [[REVERSE10]], zeroinitializer
-; AVX1-NEXT: [[TMP9:%.*]] = icmp sgt <4 x i32> [[REVERSE11]], zeroinitializer
+; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META18:![0-9]+]]
+; AVX1-NEXT: [[REVERSE:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[TMP4:%.*]] = icmp sgt <8 x i32> [[REVERSE]], zeroinitializer
; AVX1-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[IN]], i64 [[TMP0]]
-; AVX1-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP10]], i64 -3
; AVX1-NEXT: [[TMP12:%.*]] = getelementptr double, ptr [[TMP10]], i64 -7
-; AVX1-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -11
-; AVX1-NEXT: [[TMP14:%.*]] = getelementptr double, ptr [[TMP10]], i64 -15
-; AVX1-NEXT: [[REVERSE12:%.*]] = shufflevector <4 x i1> [[TMP6]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[REVERSE13:%.*]] = shufflevector <4 x i1> [[TMP7]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[REVERSE14:%.*]] = shufflevector <4 x i1> [[TMP8]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[REVERSE15:%.*]] = shufflevector <4 x i1> [[TMP9]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP11]], <4 x i1> [[REVERSE12]], <4 x double> poison), !alias.scope [[META21:![0-9]+]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP12]], <4 x i1> [[REVERSE13]], <4 x double> poison), !alias.scope [[META21]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP13]], <4 x i1> [[REVERSE14]], <4 x double> poison), !alias.scope [[META21]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP14]], <4 x i1> [[REVERSE15]], <4 x double> poison), !alias.scope [[META21]]
-; AVX1-NEXT: [[TMP15:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
-; AVX1-NEXT: [[TMP16:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
-; AVX1-NEXT: [[TMP17:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
-; AVX1-NEXT: [[TMP18:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
+; AVX1-NEXT: [[REVERSE6:%.*]] = shufflevector <8 x i1> [[TMP4]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP12]], <8 x i1> [[REVERSE6]], <8 x double> poison), !alias.scope [[META21:![0-9]+]]
+; AVX1-NEXT: [[TMP6:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
; AVX1-NEXT: [[TMP19:%.*]] = getelementptr double, ptr [[OUT]], i64 [[TMP0]]
-; AVX1-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP19]], i64 -3
; AVX1-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP19]], i64 -7
-; AVX1-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -11
-; AVX1-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP19]], i64 -15
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP15]], ptr align 8 [[TMP20]], <4 x i1> [[REVERSE12]]), !alias.scope [[META23:![0-9]+]], !noalias [[META25:![0-9]+]]
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP16]], ptr align 8 [[TMP21]], <4 x i1> [[REVERSE13]]), !alias.scope [[META23]], !noalias [[META25]]
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP17]], ptr align 8 [[TMP22]], <4 x i1> [[REVERSE14]]), !alias.scope [[META23]], !noalias [[META25]]
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP18]], ptr align 8 [[TMP23]], <4 x i1> [[REVERSE15]]), !alias.scope [[META23]], !noalias [[META25]]
-; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX1-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP6]], ptr align 8 [[TMP21]], <8 x i1> [[REVERSE6]]), !alias.scope [[META23:![0-9]+]], !noalias [[META25:![0-9]+]]
+; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; AVX1-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4096
; AVX1-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP26:![0-9]+]]
; AVX1: [[MIDDLE_BLOCK]]:
@@ -1182,49 +1125,19 @@ define void @foo6(ptr nocapture readonly %in, ptr nocapture %out, i32 %size, ptr
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = sub i64 4095, [[INDEX]]
; AVX2-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[TMP0]]
-; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -3
; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -7
-; AVX2-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -11
-; AVX2-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -15
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META22:![0-9]+]]
-; AVX2-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META22]]
-; AVX2-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META22]]
-; AVX2-NEXT: [[WIDE_LOAD8:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META22]]
-; AVX2-NEXT: [[REVERSE:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[REVERSE9:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD6]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[REVERSE10:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD7]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[REVERSE11:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD8]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[TMP6:%.*]] = icmp sgt <4 x i32> [[REVERSE]], zeroinitializer
-; AVX2-NEXT: [[TMP7:%.*]] = icmp sgt <4 x i32> [[REVERSE9]], zeroinitializer
-; AVX2-NEXT: [[TMP8:%.*]] = icmp sgt <4 x i32> [[REVERSE10]], zeroinitializer
-; AVX2-NEXT: [[TMP9:%.*]] = icmp sgt <4 x i32> [[REVERSE11]], zeroinitializer
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META22:![0-9]+]]
+; AVX2-NEXT: [[REVERSE:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[TMP4:%.*]] = icmp sgt <8 x i32> [[REVERSE]], zeroinitializer
; AVX2-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[IN]], i64 [[TMP0]]
-; AVX2-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP10]], i64 -3
; AVX2-NEXT: [[TMP12:%.*]] = getelementptr double, ptr [[TMP10]], i64 -7
-; AVX2-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -11
-; AVX2-NEXT: [[TMP14:%.*]] = getelementptr double, ptr [[TMP10]], i64 -15
-; AVX2-NEXT: [[REVERSE12:%.*]] = shufflevector <4 x i1> [[TMP6]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[REVERSE13:%.*]] = shufflevector <4 x i1> [[TMP7]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[REVERSE14:%.*]] = shufflevector <4 x i1> [[TMP8]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[REVERSE15:%.*]] = shufflevector <4 x i1> [[TMP9]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP11]], <4 x i1> [[REVERSE12]], <4 x double> poison), !alias.scope [[META25:![0-9]+]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP12]], <4 x i1> [[REVERSE13]], <4 x double> poison), !alias.scope [[META25]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP13]], <4 x i1> [[REVERSE14]], <4 x double> poison), !alias.scope [[META25]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP14]], <4 x i1> [[REVERSE15]], <4 x double> poison), !alias.scope [[META25]]
-; AVX2-NEXT: [[TMP15:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
-; AVX2-NEXT: [[TMP16:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
-; AVX2-NEXT: [[TMP17:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
-; AVX2-NEXT: [[TMP18:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
+; AVX2-NEXT: [[REVERSE6:%.*]] = shufflevector <8 x i1> [[TMP4]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP12]], <8 x i1> [[REVERSE6]], <8 x double> poison), !alias.scope [[META25:![0-9]+]]
+; AVX2-NEXT: [[TMP6:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
; AVX2-NEXT: [[TMP19:%.*]] = getelementptr double, ptr [[OUT]], i64 [[TMP0]]
-; AVX2-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP19]], i64 -3
; AVX2-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP19]], i64 -7
-; AVX2-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -11
-; AVX2-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP19]], i64 -15
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP15]], ptr align 8 [[TMP20]], <4 x i1> [[REVERSE12]]), !alias.scope [[META27:![0-9]+]], !noalias [[META29:![0-9]+]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP16]], ptr align 8 [[TMP21]], <4 x i1> [[REVERSE13]]), !alias.scope [[META27]], !noalias [[META29]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP17]], ptr align 8 [[TMP22]], <4 x i1> [[REVERSE14]]), !alias.scope [[META27]], !noalias [[META29]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP18]], ptr align 8 [[TMP23]], <4 x i1> [[REVERSE15]]), !alias.scope [[META27]], !noalias [[META29]]
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP6]], ptr align 8 [[TMP21]], <8 x i1> [[REVERSE6]]), !alias.scope [[META27:![0-9]+]], !noalias [[META29:![0-9]+]]
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; AVX2-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4096
; AVX2-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP30:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
@@ -1253,51 +1166,51 @@ define void @foo6(ptr nocapture readonly %in, ptr nocapture %out, i32 %size, ptr
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP0:%.*]] = sub i64 4095, [[INDEX]]
; AVX512-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[TMP0]]
-; AVX512-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -7
; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -15
-; AVX512-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -23
; AVX512-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -31
-; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META34:![0-9]+]]
-; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META34]]
-; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <8 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META34]]
-; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <8 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META34]]
-; AVX512-NEXT: [[REVERSE:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE9:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD6]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE10:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD7]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE11:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD8]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[TMP6:%.*]] = icmp sgt <8 x i32> [[REVERSE]], zeroinitializer
-; AVX512-NEXT: [[TMP7:%.*]] = icmp sgt <8 x i32> [[REVERSE9]], zeroinitializer
-; AVX512-NEXT: [[TMP8:%.*]] = icmp sgt <8 x i32> [[REVERSE10]], zeroinitializer
-; AVX512-NEXT: [[TMP9:%.*]] = icmp sgt <8 x i32> [[REVERSE11]], zeroinitializer
+; AVX512-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -47
+; AVX512-NEXT: [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -63
+; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META33:![0-9]+]]
+; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <16 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META33]]
+; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <16 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META33]]
+; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <16 x i32>, ptr [[TMP11]], align 4, !alias.scope [[META33]]
+; AVX512-NEXT: [[REVERSE:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE9:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD6]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE10:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD7]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE11:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD8]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[TMP6:%.*]] = icmp sgt <16 x i32> [[REVERSE]], zeroinitializer
+; AVX512-NEXT: [[TMP7:%.*]] = icmp sgt <16 x i32> [[REVERSE9]], zeroinitializer
+; AVX512-NEXT: [[TMP8:%.*]] = icmp sgt <16 x i32> [[REVERSE10]], zeroinitializer
+; AVX512-NEXT: [[TMP9:%.*]] = icmp sgt <16 x i32> [[REVERSE11]], zeroinitializer
; AVX512-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[IN]], i64 [[TMP0]]
-; AVX512-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP10]], i64 -7
; AVX512-NEXT: [[TMP12:%.*]] = getelementptr double, ptr [[TMP10]], i64 -15
-; AVX512-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -23
; AVX512-NEXT: [[TMP14:%.*]] = getelementptr double, ptr [[TMP10]], i64 -31
-; AVX512-NEXT: [[REVERSE12:%.*]] = shufflevector <8 x i1> [[TMP6]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE13:%.*]] = shufflevector <8 x i1> [[TMP7]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE14:%.*]] = shufflevector <8 x i1> [[TMP8]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE15:%.*]] = shufflevector <8 x i1> [[TMP9]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP11]], <8 x i1> [[REVERSE12]], <8 x double> poison), !alias.scope [[META37:![0-9]+]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP12]], <8 x i1> [[REVERSE13]], <8 x double> poison), !alias.scope [[META37]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP13]], <8 x i1> [[REVERSE14]], <8 x double> poison), !alias.scope [[META37]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP14]], <8 x i1> [[REVERSE15]], <8 x double> poison), !alias.scope [[META37]]
-; AVX512-NEXT: [[TMP15:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
-; AVX512-NEXT: [[TMP16:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
-; AVX512-NEXT: [[TMP17:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
-; AVX512-NEXT: [[TMP18:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -47
+; AVX512-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP10]], i64 -63
+; AVX512-NEXT: [[REVERSE12:%.*]] = shufflevector <16 x i1> [[TMP6]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE13:%.*]] = shufflevector <16 x i1> [[TMP7]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE14:%.*]] = shufflevector <16 x i1> [[TMP8]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE15:%.*]] = shufflevector <16 x i1> [[TMP9]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP12]], <16 x i1> [[REVERSE12]], <16 x double> poison), !alias.scope [[META36:![0-9]+]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP14]], <16 x i1> [[REVERSE13]], <16 x double> poison), !alias.scope [[META36]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP13]], <16 x i1> [[REVERSE14]], <16 x double> poison), !alias.scope [[META36]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP20]], <16 x i1> [[REVERSE15]], <16 x double> poison), !alias.scope [[META36]]
+; AVX512-NEXT: [[TMP15:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[TMP16:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[TMP17:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[TMP18:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
; AVX512-NEXT: [[TMP19:%.*]] = getelementptr double, ptr [[OUT]], i64 [[TMP0]]
-; AVX512-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP19]], i64 -7
; AVX512-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP19]], i64 -15
-; AVX512-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -23
; AVX512-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP19]], i64 -31
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP15]], ptr align 8 [[TMP20]], <8 x i1> [[REVERSE12]]), !alias.scope [[META39:![0-9]+]], !noalias [[META41:![0-9]+]]
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP16]], ptr align 8 [[TMP21]], <8 x i1> [[REVERSE13]]), !alias.scope [[META39]], !noalias [[META41]]
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP17]], ptr align 8 [[TMP22]], <8 x i1> [[REVERSE14]]), !alias.scope [[META39]], !noalias [[META41]]
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP18]], ptr align 8 [[TMP23]], <8 x i1> [[REVERSE15]]), !alias.scope [[META39]], !noalias [[META41]]
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX512-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -47
+; AVX512-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP19]], i64 -63
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP15]], ptr align 8 [[TMP21]], <16 x i1> [[REVERSE12]]), !alias.scope [[META38:![0-9]+]], !noalias [[META40:![0-9]+]]
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP16]], ptr align 8 [[TMP23]], <16 x i1> [[REVERSE13]]), !alias.scope [[META38]], !noalias [[META40]]
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP17]], ptr align 8 [[TMP22]], <16 x i1> [[REVERSE14]]), !alias.scope [[META38]], !noalias [[META40]]
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP18]], ptr align 8 [[TMP25]], <16 x i1> [[REVERSE15]]), !alias.scope [[META38]], !noalias [[META40]]
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
; AVX512-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4096
-; AVX512-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP42:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP41:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: br [[FOR_END:label %.*]]
; AVX512: [[SCALAR_PH]]:
@@ -1345,84 +1258,54 @@ define void @foo7(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX1-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX1: [[ITER_CHECK]]:
; AVX1-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
+; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX1: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
+; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX1: [[VECTOR_PH]]:
-; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
+; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
; AVX1-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX1-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX1: [[VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX1-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
-; AVX1-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
-; AVX1-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
-; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
-; AVX1-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
-; AVX1-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
-; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
-; AVX1-NEXT: [[TMP4:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX1-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
-; AVX1-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
-; AVX1-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX1-NEXT: [[TMP8:%.*]] = icmp ne <4 x i8> [[TMP4]], zeroinitializer
-; AVX1-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
-; AVX1-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
-; AVX1-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
+; AVX1-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX1-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
; AVX1-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX1-NEXT: [[TMP13:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
-; AVX1-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
-; AVX1-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
-; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP8]], <4 x ptr> poison)
-; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP13]], <4 x i1> [[TMP9]], <4 x ptr> poison)
-; AVX1-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
-; AVX1-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
-; AVX1-NEXT: [[TMP16:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX1-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX1-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
-; AVX1-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
-; AVX1-NEXT: [[TMP20:%.*]] = select <4 x i1> [[TMP8]], <4 x i1> [[TMP16]], <4 x i1> zeroinitializer
-; AVX1-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
-; AVX1-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
-; AVX1-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
+; AVX1-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX1-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
; AVX1-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX1-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
-; AVX1-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
-; AVX1-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP20]])
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP25]], <4 x i1> [[TMP21]])
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
-; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX1-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
+; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; AVX1-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP28:![0-9]+]]
; AVX1: [[MIDDLE_BLOCK]]:
; AVX1-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX1: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
+; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
; AVX1-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF29:![0-9]+]]
; AVX1: [[VEC_EPILOG_PH]]:
; AVX1-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
+; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
; AVX1-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX1-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX1: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_LOAD11:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
-; AVX1-NEXT: [[TMP30:%.*]] = and <4 x i8> [[WIDE_LOAD11]], splat (i8 1)
-; AVX1-NEXT: [[TMP31:%.*]] = icmp ne <4 x i8> [[TMP30]], zeroinitializer
+; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
+; AVX1-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX1-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
; AVX1-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD12:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP31]], <4 x ptr> poison)
-; AVX1-NEXT: [[TMP33:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD12]], splat (ptr null)
-; AVX1-NEXT: [[TMP34:%.*]] = select <4 x i1> [[TMP31]], <4 x i1> [[TMP33]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
+; AVX1-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX1-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
; AVX1-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP34]])
-; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
+; AVX1-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
+; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
; AVX1-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
; AVX1-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP30:![0-9]+]]
; AVX1: [[VEC_EPILOG_MIDDLE_BLOCK]]:
@@ -1437,86 +1320,56 @@ define void @foo7(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX2-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX2: [[ITER_CHECK]]:
; AVX2-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
+; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX2: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
+; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX2: [[VECTOR_PH]]:
-; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
+; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
; AVX2-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX2-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX2: [[VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
-; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
-; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
-; AVX2-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
-; AVX2-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
-; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
-; AVX2-NEXT: [[TMP4:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX2-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
-; AVX2-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
-; AVX2-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX2-NEXT: [[TMP8:%.*]] = icmp ne <4 x i8> [[TMP4]], zeroinitializer
-; AVX2-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
-; AVX2-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
-; AVX2-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
+; AVX2-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX2-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
; AVX2-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP13:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
-; AVX2-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
-; AVX2-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP8]], <4 x ptr> poison)
-; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP13]], <4 x i1> [[TMP9]], <4 x ptr> poison)
-; AVX2-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
-; AVX2-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
-; AVX2-NEXT: [[TMP16:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX2-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX2-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
-; AVX2-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
-; AVX2-NEXT: [[TMP20:%.*]] = select <4 x i1> [[TMP8]], <4 x i1> [[TMP16]], <4 x i1> zeroinitializer
-; AVX2-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
-; AVX2-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
-; AVX2-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
+; AVX2-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX2-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
; AVX2-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
-; AVX2-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
-; AVX2-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP20]])
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP25]], <4 x i1> [[TMP21]])
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX2-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; AVX2-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX2-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX2-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX2: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
-; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF33:![0-9]+]]
+; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
+; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
; AVX2: [[VEC_EPILOG_PH]]:
; AVX2-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
+; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
; AVX2-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX2-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX2: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_LOAD11:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
-; AVX2-NEXT: [[TMP30:%.*]] = and <4 x i8> [[WIDE_LOAD11]], splat (i8 1)
-; AVX2-NEXT: [[TMP31:%.*]] = icmp ne <4 x i8> [[TMP30]], zeroinitializer
+; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
+; AVX2-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX2-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
; AVX2-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD12:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP31]], <4 x ptr> poison)
-; AVX2-NEXT: [[TMP33:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD12]], splat (ptr null)
-; AVX2-NEXT: [[TMP34:%.*]] = select <4 x i1> [[TMP31]], <4 x i1> [[TMP33]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
+; AVX2-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX2-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
; AVX2-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP34]])
-; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
+; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
+; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
; AVX2-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
-; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP34:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
; AVX2: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N14:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC9]]
; AVX2-NEXT: br i1 [[CMP_N14]], [[FOR_END_LOOPEXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
@@ -1532,63 +1385,33 @@ define void @foo7(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX512-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
; AVX512-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX512: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX512-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
+; AVX512-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 64
; AVX512-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX512: [[VECTOR_PH]]:
-; AVX512-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
+; AVX512-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 63
; AVX512-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX512-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX512: [[VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
-; AVX512-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 16
-; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 24
-; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i8>, ptr [[TMP0]], align 1
-; AVX512-NEXT: [[WIDE_LOAD2:%.*]] = load <8 x i8>, ptr [[TMP1]], align 1
-; AVX512-NEXT: [[WIDE_LOAD3:%.*]] = load <8 x i8>, ptr [[TMP2]], align 1
-; AVX512-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP3]], align 1
-; AVX512-NEXT: [[TMP4:%.*]] = and <8 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX512-NEXT: [[TMP5:%.*]] = and <8 x i8> [[WIDE_LOAD2]], splat (i8 1)
-; AVX512-NEXT: [[TMP6:%.*]] = and <8 x i8> [[WIDE_LOAD3]], splat (i8 1)
-; AVX512-NEXT: [[TMP7:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX512-NEXT: [[TMP8:%.*]] = icmp ne <8 x i8> [[TMP4]], zeroinitializer
-; AVX512-NEXT: [[TMP9:%.*]] = icmp ne <8 x i8> [[TMP5]], zeroinitializer
-; AVX512-NEXT: [[TMP10:%.*]] = icmp ne <8 x i8> [[TMP6]], zeroinitializer
-; AVX512-NEXT: [[TMP11:%.*]] = icmp ne <8 x i8> [[TMP7]], zeroinitializer
+; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <64 x i8>, ptr [[TMP0]], align 1
+; AVX512-NEXT: [[TMP2:%.*]] = and <64 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX512-NEXT: [[TMP3:%.*]] = icmp ne <64 x i8> [[TMP2]], zeroinitializer
; AVX512-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP13:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
-; AVX512-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 16
-; AVX512-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 24
-; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP12]], <8 x i1> [[TMP8]], <8 x ptr> poison)
-; AVX512-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP13]], <8 x i1> [[TMP9]], <8 x ptr> poison)
-; AVX512-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP14]], <8 x i1> [[TMP10]], <8 x ptr> poison)
-; AVX512-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP15]], <8 x i1> [[TMP11]], <8 x ptr> poison)
-; AVX512-NEXT: [[TMP16:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX512-NEXT: [[TMP17:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX512-NEXT: [[TMP18:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
-; AVX512-NEXT: [[TMP19:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
-; AVX512-NEXT: [[TMP20:%.*]] = select <8 x i1> [[TMP8]], <8 x i1> [[TMP16]], <8 x i1> zeroinitializer
-; AVX512-NEXT: [[TMP21:%.*]] = select <8 x i1> [[TMP9]], <8 x i1> [[TMP17]], <8 x i1> zeroinitializer
-; AVX512-NEXT: [[TMP22:%.*]] = select <8 x i1> [[TMP10]], <8 x i1> [[TMP18]], <8 x i1> zeroinitializer
-; AVX512-NEXT: [[TMP23:%.*]] = select <8 x i1> [[TMP11]], <8 x i1> [[TMP19]], <8 x i1> zeroinitializer
+; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <64 x ptr> @llvm.masked.load.v64p0.p0(ptr align 8 [[TMP12]], <64 x i1> [[TMP3]], <64 x ptr> poison)
+; AVX512-NEXT: [[TMP5:%.*]] = icmp ne <64 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX512-NEXT: [[TMP6:%.*]] = select <64 x i1> [[TMP3]], <64 x i1> [[TMP5]], <64 x i1> zeroinitializer
; AVX512-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
-; AVX512-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 16
-; AVX512-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 24
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <8 x i1> [[TMP20]])
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP25]], <8 x i1> [[TMP21]])
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <8 x i1> [[TMP22]])
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <8 x i1> [[TMP23]])
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX512-NEXT: call void @llvm.masked.store.v64f64.p0(<64 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <64 x i1> [[TMP6]])
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
; AVX512-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; AVX512-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP44:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP43:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX512-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX512: [[VEC_EPILOG_ITER_CHECK]]:
; AVX512-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
-; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF21]]
+; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF44:![0-9]+]]
; AVX512: [[VEC_EPILOG_PH]]:
; AVX512-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX512-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
@@ -1666,84 +1489,54 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX1-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX1: [[ITER_CHECK]]:
; AVX1-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
+; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX1: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
+; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX1: [[VECTOR_PH]]:
-; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
+; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
; AVX1-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX1-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX1: [[VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX1-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
-; AVX1-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
-; AVX1-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
-; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
-; AVX1-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
-; AVX1-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
-; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
-; AVX1-NEXT: [[TMP4:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX1-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
-; AVX1-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
-; AVX1-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX1-NEXT: [[TMP8:%.*]] = icmp ne <4 x i8> [[TMP4]], zeroinitializer
-; AVX1-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
-; AVX1-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
-; AVX1-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
+; AVX1-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX1-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
; AVX1-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX1-NEXT: [[TMP13:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
-; AVX1-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
-; AVX1-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
-; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP8]], <4 x ptr> poison)
-; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP13]], <4 x i1> [[TMP9]], <4 x ptr> poison)
-; AVX1-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
-; AVX1-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
-; AVX1-NEXT: [[TMP16:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX1-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX1-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
-; AVX1-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
-; AVX1-NEXT: [[TMP20:%.*]] = select <4 x i1> [[TMP8]], <4 x i1> [[TMP16]], <4 x i1> zeroinitializer
-; AVX1-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
-; AVX1-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
-; AVX1-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
+; AVX1-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX1-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
; AVX1-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX1-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
-; AVX1-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
-; AVX1-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP20]])
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP25]], <4 x i1> [[TMP21]])
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
-; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX1-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
+; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; AVX1-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
; AVX1: [[MIDDLE_BLOCK]]:
; AVX1-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX1: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
+; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
; AVX1-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF29]]
; AVX1: [[VEC_EPILOG_PH]]:
; AVX1-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
+; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
; AVX1-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX1-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX1: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_LOAD11:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
-; AVX1-NEXT: [[TMP30:%.*]] = and <4 x i8> [[WIDE_LOAD11]], splat (i8 1)
-; AVX1-NEXT: [[TMP31:%.*]] = icmp ne <4 x i8> [[TMP30]], zeroinitializer
+; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
+; AVX1-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX1-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
; AVX1-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD12:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP31]], <4 x ptr> poison)
-; AVX1-NEXT: [[TMP33:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD12]], splat (ptr null)
-; AVX1-NEXT: [[TMP34:%.*]] = select <4 x i1> [[TMP31]], <4 x i1> [[TMP33]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
+; AVX1-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX1-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
; AVX1-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP34]])
-; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
+; AVX1-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
+; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
; AVX1-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
; AVX1-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
; AVX1: [[VEC_EPILOG_MIDDLE_BLOCK]]:
@@ -1758,86 +1551,56 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX2-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX2: [[ITER_CHECK]]:
; AVX2-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
+; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX2: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
+; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX2: [[VECTOR_PH]]:
-; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
+; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
; AVX2-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX2-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX2: [[VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
-; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
-; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
-; AVX2-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
-; AVX2-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
-; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
-; AVX2-NEXT: [[TMP4:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX2-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
-; AVX2-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
-; AVX2-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX2-NEXT: [[TMP8:%.*]] = icmp ne <4 x i8> [[TMP4]], zeroinitializer
-; AVX2-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
-; AVX2-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
-; AVX2-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
+; AVX2-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX2-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
; AVX2-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP13:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
-; AVX2-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
-; AVX2-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP8]], <4 x ptr> poison)
-; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP13]], <4 x i1> [[TMP9]], <4 x ptr> poison)
-; AVX2-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
-; AVX2-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
-; AVX2-NEXT: [[TMP16:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX2-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX2-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
-; AVX2-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
-; AVX2-NEXT: [[TMP20:%.*]] = select <4 x i1> [[TMP8]], <4 x i1> [[TMP16]], <4 x i1> zeroinitializer
-; AVX2-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
-; AVX2-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
-; AVX2-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
+; AVX2-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX2-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
; AVX2-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
-; AVX2-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
-; AVX2-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP20]])
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP25]], <4 x i1> [[TMP21]])
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX2-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; AVX2-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; AVX2-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP35:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX2-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX2: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
-; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF33]]
+; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
+; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
; AVX2: [[VEC_EPILOG_PH]]:
; AVX2-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
+; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
; AVX2-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX2-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX2: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_LOAD11:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
-; AVX2-NEXT: [[TMP30:%.*]] = and <4 x i8> [[WIDE_LOAD11]], splat (i8 1)
-; AVX2-NEXT: [[TMP31:%.*]] = icmp ne <4 x i8> [[TMP30]], zeroinitializer
+; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
+; AVX2-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX2-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
; AVX2-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD12:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP31]], <4 x ptr> poison)
-; AVX2-NEXT: [[TMP33:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD12]], splat (ptr null)
-; AVX2-NEXT: [[TMP34:%.*]] = select <4 x i1> [[TMP31]], <4 x i1> [[TMP33]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
+; AVX2-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX2-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
; AVX2-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP34]])
-; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
+; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
+; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
; AVX2-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
-; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP37:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
; AVX2: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N14:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC9]]
; AVX2-NEXT: br i1 [[CMP_N14]], [[FOR_END_LOOPEXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
@@ -1853,55 +1616,25 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX512-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
; AVX512-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX512: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX512-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
+; AVX512-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 64
; AVX512-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX512: [[VECTOR_PH]]:
-; AVX512-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
+; AVX512-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 63
; AVX512-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX512-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX512: [[VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
-; AVX512-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 16
-; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 24
-; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i8>, ptr [[TMP0]], align 1
-; AVX512-NEXT: [[WIDE_LOAD2:%.*]] = load <8 x i8>, ptr [[TMP1]], align 1
-; AVX512-NEXT: [[WIDE_LOAD3:%.*]] = load <8 x i8>, ptr [[TMP2]], align 1
-; AVX512-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP3]], align 1
-; AVX512-NEXT: [[TMP4:%.*]] = and <8 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX512-NEXT: [[TMP5:%.*]] = and <8 x i8> [[WIDE_LOAD2]], splat (i8 1)
-; AVX512-NEXT: [[TMP6:%.*]] = and <8 x i8> [[WIDE_LOAD3]], splat (i8 1)
-; AVX512-NEXT: [[TMP7:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX512-NEXT: [[TMP8:%.*]] = icmp ne <8 x i8> [[TMP4]], zeroinitializer
-; AVX512-NEXT: [[TMP9:%.*]] = icmp ne <8 x i8> [[TMP5]], zeroinitializer
-; AVX512-NEXT: [[TMP10:%.*]] = icmp ne <8 x i8> [[TMP6]], zeroinitializer
-; AVX512-NEXT: [[TMP11:%.*]] = icmp ne <8 x i8> [[TMP7]], zeroinitializer
+; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <64 x i8>, ptr [[TMP0]], align 1
+; AVX512-NEXT: [[TMP2:%.*]] = and <64 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX512-NEXT: [[TMP3:%.*]] = icmp ne <64 x i8> [[TMP2]], zeroinitializer
; AVX512-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP13:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
-; AVX512-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 16
-; AVX512-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 24
-; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP12]], <8 x i1> [[TMP8]], <8 x ptr> poison)
-; AVX512-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP13]], <8 x i1> [[TMP9]], <8 x ptr> poison)
-; AVX512-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP14]], <8 x i1> [[TMP10]], <8 x ptr> poison)
-; AVX512-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP15]], <8 x i1> [[TMP11]], <8 x ptr> poison)
-; AVX512-NEXT: [[TMP16:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX512-NEXT: [[TMP17:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX512-NEXT: [[TMP18:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
-; AVX512-NEXT: [[TMP19:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
-; AVX512-NEXT: [[TMP20:%.*]] = select <8 x i1> [[TMP8]], <8 x i1> [[TMP16]], <8 x i1> zeroinitializer
-; AVX512-NEXT: [[TMP21:%.*]] = select <8 x i1> [[TMP9]], <8 x i1> [[TMP17]], <8 x i1> zeroinitializer
-; AVX512-NEXT: [[TMP22:%.*]] = select <8 x i1> [[TMP10]], <8 x i1> [[TMP18]], <8 x i1> zeroinitializer
-; AVX512-NEXT: [[TMP23:%.*]] = select <8 x i1> [[TMP11]], <8 x i1> [[TMP19]], <8 x i1> zeroinitializer
+; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <64 x ptr> @llvm.masked.load.v64p0.p0(ptr align 8 [[TMP12]], <64 x i1> [[TMP3]], <64 x ptr> poison)
+; AVX512-NEXT: [[TMP5:%.*]] = icmp ne <64 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX512-NEXT: [[TMP6:%.*]] = select <64 x i1> [[TMP3]], <64 x i1> [[TMP5]], <64 x i1> zeroinitializer
; AVX512-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
-; AVX512-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 16
-; AVX512-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 24
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <8 x i1> [[TMP20]])
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP25]], <8 x i1> [[TMP21]])
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <8 x i1> [[TMP22]])
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <8 x i1> [[TMP23]])
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX512-NEXT: call void @llvm.masked.store.v64f64.p0(<64 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <64 x i1> [[TMP6]])
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
; AVX512-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX512-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP47:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
@@ -1909,7 +1642,7 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX512-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX512: [[VEC_EPILOG_ITER_CHECK]]:
; AVX512-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
-; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF21]]
+; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF44]]
; AVX512: [[VEC_EPILOG_PH]]:
; AVX512-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX512-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
@@ -2002,10 +1735,10 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX2: [[VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; AVX2-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP100:%.*]], %[[VECTOR_BODY]] ]
-; AVX2-NEXT: [[VEC_PHI1:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP101:%.*]], %[[VECTOR_BODY]] ]
-; AVX2-NEXT: [[VEC_PHI2:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP102:%.*]], %[[VECTOR_BODY]] ]
-; AVX2-NEXT: [[VEC_PHI3:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP103:%.*]], %[[VECTOR_BODY]] ]
+; AVX2-NEXT: [[VEC_PHI:%.*]] = phi <8 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP196:%.*]], %[[VECTOR_BODY]] ]
+; AVX2-NEXT: [[VEC_PHI1:%.*]] = phi <8 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP197:%.*]], %[[VECTOR_BODY]] ]
+; AVX2-NEXT: [[VEC_PHI2:%.*]] = phi <8 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP198:%.*]], %[[VECTOR_BODY]] ]
+; AVX2-NEXT: [[VEC_PHI3:%.*]] = phi <8 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP199:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = sub i32 100, [[INDEX]]
; AVX2-NEXT: [[TMP1:%.*]] = add i32 [[TMP0]], -1
; AVX2-NEXT: [[TMP2:%.*]] = add i32 [[TMP0]], -2
@@ -2022,6 +1755,22 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: [[TMP13:%.*]] = add i32 [[TMP0]], -13
; AVX2-NEXT: [[TMP14:%.*]] = add i32 [[TMP0]], -14
; AVX2-NEXT: [[TMP15:%.*]] = add i32 [[TMP0]], -15
+; AVX2-NEXT: [[TMP68:%.*]] = add i32 [[TMP0]], -16
+; AVX2-NEXT: [[TMP69:%.*]] = add i32 [[TMP0]], -17
+; AVX2-NEXT: [[TMP70:%.*]] = add i32 [[TMP0]], -18
+; AVX2-NEXT: [[TMP71:%.*]] = add i32 [[TMP0]], -19
+; AVX2-NEXT: [[TMP76:%.*]] = add i32 [[TMP0]], -20
+; AVX2-NEXT: [[TMP77:%.*]] = add i32 [[TMP0]], -21
+; AVX2-NEXT: [[TMP78:%.*]] = add i32 [[TMP0]], -22
+; AVX2-NEXT: [[TMP79:%.*]] = add i32 [[TMP0]], -23
+; AVX2-NEXT: [[TMP96:%.*]] = add i32 [[TMP0]], -24
+; AVX2-NEXT: [[TMP97:%.*]] = add i32 [[TMP0]], -25
+; AVX2-NEXT: [[TMP98:%.*]] = add i32 [[TMP0]], -26
+; AVX2-NEXT: [[TMP99:%.*]] = add i32 [[TMP0]], -27
+; AVX2-NEXT: [[TMP100:%.*]] = add i32 [[TMP0]], -28
+; AVX2-NEXT: [[TMP101:%.*]] = add i32 [[TMP0]], -29
+; AVX2-NEXT: [[TMP102:%.*]] = add i32 [[TMP0]], -30
+; AVX2-NEXT: [[TMP103:%.*]] = add i32 [[TMP0]], -31
; AVX2-NEXT: [[TMP16:%.*]] = zext i32 [[TMP0]] to i64
; AVX2-NEXT: [[TMP17:%.*]] = zext i32 [[TMP1]] to i64
; AVX2-NEXT: [[TMP18:%.*]] = zext i32 [[TMP2]] to i64
@@ -2038,6 +1787,22 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: [[TMP29:%.*]] = zext i32 [[TMP13]] to i64
; AVX2-NEXT: [[TMP30:%.*]] = zext i32 [[TMP14]] to i64
; AVX2-NEXT: [[TMP31:%.*]] = zext i32 [[TMP15]] to i64
+; AVX2-NEXT: [[TMP144:%.*]] = zext i32 [[TMP68]] to i64
+; AVX2-NEXT: [[TMP145:%.*]] = zext i32 [[TMP69]] to i64
+; AVX2-NEXT: [[TMP146:%.*]] = zext i32 [[TMP70]] to i64
+; AVX2-NEXT: [[TMP147:%.*]] = zext i32 [[TMP71]] to i64
+; AVX2-NEXT: [[TMP148:%.*]] = zext i32 [[TMP76]] to i64
+; AVX2-NEXT: [[TMP149:%.*]] = zext i32 [[TMP77]] to i64
+; AVX2-NEXT: [[TMP150:%.*]] = zext i32 [[TMP78]] to i64
+; AVX2-NEXT: [[TMP151:%.*]] = zext i32 [[TMP79]] to i64
+; AVX2-NEXT: [[TMP200:%.*]] = zext i32 [[TMP96]] to i64
+; AVX2-NEXT: [[TMP201:%.*]] = zext i32 [[TMP97]] to i64
+; AVX2-NEXT: [[TMP202:%.*]] = zext i32 [[TMP98]] to i64
+; AVX2-NEXT: [[TMP203:%.*]] = zext i32 [[TMP99]] to i64
+; AVX2-NEXT: [[TMP204:%.*]] = zext i32 [[TMP100]] to i64
+; AVX2-NEXT: [[TMP205:%.*]] = zext i32 [[TMP101]] to i64
+; AVX2-NEXT: [[TMP206:%.*]] = zext i32 [[TMP102]] to i64
+; AVX2-NEXT: [[TMP207:%.*]] = zext i32 [[TMP103]] to i64
; AVX2-NEXT: [[TMP32:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP16]]
; AVX2-NEXT: [[TMP33:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP17]]
; AVX2-NEXT: [[TMP34:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP18]]
@@ -2054,6 +1819,22 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: [[TMP45:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP29]]
; AVX2-NEXT: [[TMP46:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP30]]
; AVX2-NEXT: [[TMP47:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP31]]
+; AVX2-NEXT: [[TMP208:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP144]]
+; AVX2-NEXT: [[TMP209:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP145]]
+; AVX2-NEXT: [[TMP210:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP146]]
+; AVX2-NEXT: [[TMP211:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP147]]
+; AVX2-NEXT: [[TMP84:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP148]]
+; AVX2-NEXT: [[TMP85:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP149]]
+; AVX2-NEXT: [[TMP86:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP150]]
+; AVX2-NEXT: [[TMP87:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP151]]
+; AVX2-NEXT: [[TMP212:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP200]]
+; AVX2-NEXT: [[TMP213:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP201]]
+; AVX2-NEXT: [[TMP214:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP202]]
+; AVX2-NEXT: [[TMP215:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP203]]
+; AVX2-NEXT: [[TMP92:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP204]]
+; AVX2-NEXT: [[TMP93:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP205]]
+; AVX2-NEXT: [[TMP94:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP206]]
+; AVX2-NEXT: [[TMP95:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP207]]
; AVX2-NEXT: [[TMP48:%.*]] = load ptr, ptr [[TMP32]], align 8
; AVX2-NEXT: [[TMP49:%.*]] = load ptr, ptr [[TMP33]], align 8
; AVX2-NEXT: [[TMP50:%.*]] = load ptr, ptr [[TMP34]], align 8
@@ -2070,59 +1851,107 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: [[TMP61:%.*]] = load ptr, ptr [[TMP45]], align 8
; AVX2-NEXT: [[TMP62:%.*]] = load ptr, ptr [[TMP46]], align 8
; AVX2-NEXT: [[TMP63:%.*]] = load ptr, ptr [[TMP47]], align 8
+; AVX2-NEXT: [[TMP216:%.*]] = load ptr, ptr [[TMP208]], align 8
+; AVX2-NEXT: [[TMP217:%.*]] = load ptr, ptr [[TMP209]], align 8
+; AVX2-NEXT: [[TMP218:%.*]] = load ptr, ptr [[TMP210]], align 8
+; AVX2-NEXT: [[TMP219:%.*]] = load ptr, ptr [[TMP211]], align 8
+; AVX2-NEXT: [[TMP220:%.*]] = load ptr, ptr [[TMP84]], align 8
+; AVX2-NEXT: [[TMP221:%.*]] = load ptr, ptr [[TMP85]], align 8
+; AVX2-NEXT: [[TMP222:%.*]] = load ptr, ptr [[TMP86]], align 8
+; AVX2-NEXT: [[TMP223:%.*]] = load ptr, ptr [[TMP87]], align 8
+; AVX2-NEXT: [[TMP224:%.*]] = load ptr, ptr [[TMP212]], align 8
+; AVX2-NEXT: [[TMP225:%.*]] = load ptr, ptr [[TMP213]], align 8
+; AVX2-NEXT: [[TMP226:%.*]] = load ptr, ptr [[TMP214]], align 8
+; AVX2-NEXT: [[TMP227:%.*]] = load ptr, ptr [[TMP215]], align 8
+; AVX2-NEXT: [[TMP228:%.*]] = load ptr, ptr [[TMP92]], align 8
+; AVX2-NEXT: [[TMP229:%.*]] = load ptr, ptr [[TMP93]], align 8
+; AVX2-NEXT: [[TMP230:%.*]] = load ptr, ptr [[TMP94]], align 8
+; AVX2-NEXT: [[TMP231:%.*]] = load ptr, ptr [[TMP95]], align 8
; AVX2-NEXT: [[TMP64:%.*]] = load i32, ptr [[TMP48]], align 8
; AVX2-NEXT: [[TMP65:%.*]] = load i32, ptr [[TMP49]], align 8
; AVX2-NEXT: [[TMP66:%.*]] = load i32, ptr [[TMP50]], align 8
; AVX2-NEXT: [[TMP67:%.*]] = load i32, ptr [[TMP51]], align 8
-; AVX2-NEXT: [[TMP68:%.*]] = insertelement <4 x i32> poison, i32 [[TMP64]], i64 0
-; AVX2-NEXT: [[TMP69:%.*]] = insertelement <4 x i32> [[TMP68]], i32 [[TMP65]], i64 1
-; AVX2-NEXT: [[TMP70:%.*]] = insertelement <4 x i32> [[TMP69]], i32 [[TMP66]], i64 2
-; AVX2-NEXT: [[TMP71:%.*]] = insertelement <4 x i32> [[TMP70]], i32 [[TMP67]], i64 3
; AVX2-NEXT: [[TMP72:%.*]] = load i32, ptr [[TMP52]], align 8
; AVX2-NEXT: [[TMP73:%.*]] = load i32, ptr [[TMP53]], align 8
; AVX2-NEXT: [[TMP74:%.*]] = load i32, ptr [[TMP54]], align 8
; AVX2-NEXT: [[TMP75:%.*]] = load i32, ptr [[TMP55]], align 8
-; AVX2-NEXT: [[TMP76:%.*]] = insertelement <4 x i32> poison, i32 [[TMP72]], i64 0
-; AVX2-NEXT: [[TMP77:%.*]] = insertelement <4 x i32> [[TMP76]], i32 [[TMP73]], i64 1
-; AVX2-NEXT: [[TMP78:%.*]] = insertelement <4 x i32> [[TMP77]], i32 [[TMP74]], i64 2
-; AVX2-NEXT: [[TMP79:%.*]] = insertelement <4 x i32> [[TMP78]], i32 [[TMP75]], i64 3
+; AVX2-NEXT: [[TMP232:%.*]] = insertelement <8 x i32> poison, i32 [[TMP64]], i64 0
+; AVX2-NEXT: [[TMP137:%.*]] = insertelement <8 x i32> [[TMP232]], i32 [[TMP65]], i64 1
+; AVX2-NEXT: [[TMP138:%.*]] = insertelement <8 x i32> [[TMP137]], i32 [[TMP66]], i64 2
+; AVX2-NEXT: [[TMP139:%.*]] = insertelement <8 x i32> [[TMP138]], i32 [[TMP67]], i64 3
+; AVX2-NEXT: [[TMP140:%.*]] = insertelement <8 x i32> [[TMP139]], i32 [[TMP72]], i64 4
+; AVX2-NEXT: [[TMP141:%.*]] = insertelement <8 x i32> [[TMP140]], i32 [[TMP73]], i64 5
+; AVX2-NEXT: [[TMP142:%.*]] = insertelement <8 x i32> [[TMP141]], i32 [[TMP74]], i64 6
+; AVX2-NEXT: [[TMP143:%.*]] = insertelement <8 x i32> [[TMP142]], i32 [[TMP75]], i64 7
; AVX2-NEXT: [[TMP80:%.*]] = load i32, ptr [[TMP56]], align 8
; AVX2-NEXT: [[TMP81:%.*]] = load i32, ptr [[TMP57]], align 8
; AVX2-NEXT: [[TMP82:%.*]] = load i32, ptr [[TMP58]], align 8
; AVX2-NEXT: [[TMP83:%.*]] = load i32, ptr [[TMP59]], align 8
-; AVX2-NEXT: [[TMP84:%.*]] = insertelement <4 x i32> poison, i32 [[TMP80]], i64 0
-; AVX2-NEXT: [[TMP85:%.*]] = insertelement <4 x i32> [[TMP84]], i32 [[TMP81]], i64 1
-; AVX2-NEXT: [[TMP86:%.*]] = insertelement <4 x i32> [[TMP85]], i32 [[TMP82]], i64 2
-; AVX2-NEXT: [[TMP87:%.*]] = insertelement <4 x i32> [[TMP86]], i32 [[TMP83]], i64 3
; AVX2-NEXT: [[TMP88:%.*]] = load i32, ptr [[TMP60]], align 8
; AVX2-NEXT: [[TMP89:%.*]] = load i32, ptr [[TMP61]], align 8
; AVX2-NEXT: [[TMP90:%.*]] = load i32, ptr [[TMP62]], align 8
; AVX2-NEXT: [[TMP91:%.*]] = load i32, ptr [[TMP63]], align 8
-; AVX2-NEXT: [[TMP92:%.*]] = insertelement <4 x i32> poison, i32 [[TMP88]], i64 0
-; AVX2-NEXT: [[TMP93:%.*]] = insertelement <4 x i32> [[TMP92]], i32 [[TMP89]], i64 1
-; AVX2-NEXT: [[TMP94:%.*]] = insertelement <4 x i32> [[TMP93]], i32 [[TMP90]], i64 2
-; AVX2-NEXT: [[TMP95:%.*]] = insertelement <4 x i32> [[TMP94]], i32 [[TMP91]], i64 3
-; AVX2-NEXT: [[TMP96:%.*]] = icmp ne <4 x i32> [[TMP71]], zeroinitializer
-; AVX2-NEXT: [[TMP97:%.*]] = icmp ne <4 x i32> [[TMP79]], zeroinitializer
-; AVX2-NEXT: [[TMP98:%.*]] = icmp ne <4 x i32> [[TMP87]], zeroinitializer
-; AVX2-NEXT: [[TMP99:%.*]] = icmp ne <4 x i32> [[TMP95]], zeroinitializer
-; AVX2-NEXT: [[TMP100]] = or <4 x i1> [[VEC_PHI]], [[TMP96]]
-; AVX2-NEXT: [[TMP101]] = or <4 x i1> [[VEC_PHI1]], [[TMP97]]
-; AVX2-NEXT: [[TMP102]] = or <4 x i1> [[VEC_PHI2]], [[TMP98]]
-; AVX2-NEXT: [[TMP103]] = or <4 x i1> [[VEC_PHI3]], [[TMP99]]
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 16
+; AVX2-NEXT: [[TMP152:%.*]] = insertelement <8 x i32> poison, i32 [[TMP80]], i64 0
+; AVX2-NEXT: [[TMP153:%.*]] = insertelement <8 x i32> [[TMP152]], i32 [[TMP81]], i64 1
+; AVX2-NEXT: [[TMP154:%.*]] = insertelement <8 x i32> [[TMP153]], i32 [[TMP82]], i64 2
+; AVX2-NEXT: [[TMP155:%.*]] = insertelement <8 x i32> [[TMP154]], i32 [[TMP83]], i64 3
+; AVX2-NEXT: [[TMP156:%.*]] = insertelement <8 x i32> [[TMP155]], i32 [[TMP88]], i64 4
+; AVX2-NEXT: [[TMP157:%.*]] = insertelement <8 x i32> [[TMP156]], i32 [[TMP89]], i64 5
+; AVX2-NEXT: [[TMP158:%.*]] = insertelement <8 x i32> [[TMP157]], i32 [[TMP90]], i64 6
+; AVX2-NEXT: [[TMP159:%.*]] = insertelement <8 x i32> [[TMP158]], i32 [[TMP91]], i64 7
+; AVX2-NEXT: [[TMP160:%.*]] = load i32, ptr [[TMP216]], align 8
+; AVX2-NEXT: [[TMP161:%.*]] = load i32, ptr [[TMP217]], align 8
+; AVX2-NEXT: [[TMP162:%.*]] = load i32, ptr [[TMP218]], align 8
+; AVX2-NEXT: [[TMP163:%.*]] = load i32, ptr [[TMP219]], align 8
+; AVX2-NEXT: [[TMP164:%.*]] = load i32, ptr [[TMP220]], align 8
+; AVX2-NEXT: [[TMP165:%.*]] = load i32, ptr [[TMP221]], align 8
+; AVX2-NEXT: [[TMP166:%.*]] = load i32, ptr [[TMP222]], align 8
+; AVX2-NEXT: [[TMP167:%.*]] = load i32, ptr [[TMP223]], align 8
+; AVX2-NEXT: [[TMP168:%.*]] = insertelement <8 x i32> poison, i32 [[TMP160]], i64 0
+; AVX2-NEXT: [[TMP169:%.*]] = insertelement <8 x i32> [[TMP168]], i32 [[TMP161]], i64 1
+; AVX2-NEXT: [[TMP170:%.*]] = insertelement <8 x i32> [[TMP169]], i32 [[TMP162]], i64 2
+; AVX2-NEXT: [[TMP171:%.*]] = insertelement <8 x i32> [[TMP170]], i32 [[TMP163]], i64 3
+; AVX2-NEXT: [[TMP172:%.*]] = insertelement <8 x i32> [[TMP171]], i32 [[TMP164]], i64 4
+; AVX2-NEXT: [[TMP173:%.*]] = insertelement <8 x i32> [[TMP172]], i32 [[TMP165]], i64 5
+; AVX2-NEXT: [[TMP174:%.*]] = insertelement <8 x i32> [[TMP173]], i32 [[TMP166]], i64 6
+; AVX2-NEXT: [[TMP175:%.*]] = insertelement <8 x i32> [[TMP174]], i32 [[TMP167]], i64 7
+; AVX2-NEXT: [[TMP176:%.*]] = load i32, ptr [[TMP224]], align 8
+; AVX2-NEXT: [[TMP177:%.*]] = load i32, ptr [[TMP225]], align 8
+; AVX2-NEXT: [[TMP178:%.*]] = load i32, ptr [[TMP226]], align 8
+; AVX2-NEXT: [[TMP179:%.*]] = load i32, ptr [[TMP227]], align 8
+; AVX2-NEXT: [[TMP180:%.*]] = load i32, ptr [[TMP228]], align 8
+; AVX2-NEXT: [[TMP181:%.*]] = load i32, ptr [[TMP229]], align 8
+; AVX2-NEXT: [[TMP182:%.*]] = load i32, ptr [[TMP230]], align 8
+; AVX2-NEXT: [[TMP183:%.*]] = load i32, ptr [[TMP231]], align 8
+; AVX2-NEXT: [[TMP184:%.*]] = insertelement <8 x i32> poison, i32 [[TMP176]], i64 0
+; AVX2-NEXT: [[TMP185:%.*]] = insertelement <8 x i32> [[TMP184]], i32 [[TMP177]], i64 1
+; AVX2-NEXT: [[TMP186:%.*]] = insertelement <8 x i32> [[TMP185]], i32 [[TMP178]], i64 2
+; AVX2-NEXT: [[TMP187:%.*]] = insertelement <8 x i32> [[TMP186]], i32 [[TMP179]], i64 3
+; AVX2-NEXT: [[TMP188:%.*]] = insertelement <8 x i32> [[TMP187]], i32 [[TMP180]], i64 4
+; AVX2-NEXT: [[TMP189:%.*]] = insertelement <8 x i32> [[TMP188]], i32 [[TMP181]], i64 5
+; AVX2-NEXT: [[TMP190:%.*]] = insertelement <8 x i32> [[TMP189]], i32 [[TMP182]], i64 6
+; AVX2-NEXT: [[TMP191:%.*]] = insertelement <8 x i32> [[TMP190]], i32 [[TMP183]], i64 7
+; AVX2-NEXT: [[TMP192:%.*]] = icmp ne <8 x i32> [[TMP143]], zeroinitializer
+; AVX2-NEXT: [[TMP193:%.*]] = icmp ne <8 x i32> [[TMP159]], zeroinitializer
+; AVX2-NEXT: [[TMP194:%.*]] = icmp ne <8 x i32> [[TMP175]], zeroinitializer
+; AVX2-NEXT: [[TMP195:%.*]] = icmp ne <8 x i32> [[TMP191]], zeroinitializer
+; AVX2-NEXT: [[TMP196]] = or <8 x i1> [[VEC_PHI]], [[TMP192]]
+; AVX2-NEXT: [[TMP197]] = or <8 x i1> [[VEC_PHI1]], [[TMP193]]
+; AVX2-NEXT: [[TMP198]] = or <8 x i1> [[VEC_PHI2]], [[TMP194]]
+; AVX2-NEXT: [[TMP199]] = or <8 x i1> [[VEC_PHI3]], [[TMP195]]
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 32
; AVX2-NEXT: [[TMP104:%.*]] = icmp eq i32 [[INDEX_NEXT]], 96
-; AVX2-NEXT: br i1 [[TMP104]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP39:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP104]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP38:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
-; AVX2-NEXT: [[BIN_RDX:%.*]] = or <4 x i1> [[TMP101]], [[TMP100]]
-; AVX2-NEXT: [[BIN_RDX4:%.*]] = or <4 x i1> [[TMP102]], [[BIN_RDX]]
-; AVX2-NEXT: [[BIN_RDX5:%.*]] = or <4 x i1> [[TMP103]], [[BIN_RDX4]]
-; AVX2-NEXT: [[TMP105:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[BIN_RDX5]])
+; AVX2-NEXT: [[BIN_RDX:%.*]] = or <8 x i1> [[TMP197]], [[TMP196]]
+; AVX2-NEXT: [[BIN_RDX4:%.*]] = or <8 x i1> [[TMP198]], [[BIN_RDX]]
+; AVX2-NEXT: [[BIN_RDX5:%.*]] = or <8 x i1> [[TMP199]], [[BIN_RDX4]]
+; AVX2-NEXT: [[TMP105:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[BIN_RDX5]])
; AVX2-NEXT: [[TMP106:%.*]] = freeze i1 [[TMP105]]
; AVX2-NEXT: [[RDX_SELECT:%.*]] = select i1 [[TMP106]], i32 0, i32 1
; AVX2-NEXT: br i1 false, [[EXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX2: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX2-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF33]]
+; AVX2-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF39:![0-9]+]]
; AVX2: [[VEC_EPILOG_PH]]:
; AVX2-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ 96, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX2-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 1, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
diff --git a/llvm/test/Transforms/LoopVectorize/X86/maxbw-cast-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/maxbw-cast-cost.ll
new file mode 100644
index 0000000000000..620cadf0e5012
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/X86/maxbw-cast-cost.ll
@@ -0,0 +1,353 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -force-vector-interleave=1 \
+; RUN: -mcpu=znver4 -S %s | FileCheck %s --check-prefix=MAXBW
+
+target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-linux-gnu"
+
+; i8 x i8 dot product accumulated to i32. Smallest type is i8, widest is
+; i32. With MaxBW enabled, the VF is chosen from the smallest type, so
+; MaxBW should choose VF=64 (512/8) rather than the natural VF=16 (512/32).
+define void @dot_i8(ptr noalias %weights, ptr noalias %input, ptr noalias %output, i64 %n) {
+; MAXBW-LABEL: define void @dot_i8(
+; MAXBW-SAME: ptr noalias [[WEIGHTS:%.*]], ptr noalias [[INPUT:%.*]], ptr noalias [[OUTPUT:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; MAXBW-NEXT: [[ITER_CHECK:.*]]:
+; MAXBW-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; MAXBW-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; MAXBW: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; MAXBW-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 64
+; MAXBW-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; MAXBW: [[VECTOR_PH]]:
+; MAXBW-NEXT: [[N_MOD_VF:%.*]] = and i64 [[N]], 63
+; MAXBW-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; MAXBW-NEXT: br label %[[VECTOR_BODY:.*]]
+; MAXBW: [[VECTOR_BODY]]:
+; MAXBW-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; MAXBW-NEXT: [[VEC_PHI:%.*]] = phi <64 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP7:%.*]], %[[VECTOR_BODY]] ]
+; MAXBW-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[WEIGHTS]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD:%.*]] = load <64 x i8>, ptr [[TMP0]], align 1
+; MAXBW-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[INPUT]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD2:%.*]] = load <64 x i8>, ptr [[TMP1]], align 1
+; MAXBW-NEXT: [[TMP2:%.*]] = sext <64 x i8> [[WIDE_LOAD]] to <64 x i32>
+; MAXBW-NEXT: [[TMP3:%.*]] = zext <64 x i8> [[WIDE_LOAD2]] to <64 x i32>
+; MAXBW-NEXT: [[TMP4:%.*]] = mul nsw <64 x i32> [[TMP2]], [[TMP3]]
+; MAXBW-NEXT: [[TMP7]] = add <64 x i32> [[TMP4]], [[VEC_PHI]]
+; MAXBW-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
+; MAXBW-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; MAXBW-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; MAXBW: [[MIDDLE_BLOCK]]:
+; MAXBW-NEXT: [[TMP8:%.*]] = call i32 @llvm.vector.reduce.add.v64i32(<64 x i32> [[TMP7]])
+; MAXBW-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; MAXBW-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; MAXBW: [[VEC_EPILOG_ITER_CHECK]]:
+; MAXBW-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
+; MAXBW-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3:![0-9]+]]
+; MAXBW: [[VEC_EPILOG_PH]]:
+; MAXBW-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; MAXBW-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP8]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; MAXBW-NEXT: [[N_MOD_VF3:%.*]] = and i64 [[N]], 7
+; MAXBW-NEXT: [[N_VEC4:%.*]] = sub i64 [[N]], [[N_MOD_VF3]]
+; MAXBW-NEXT: [[TMP17:%.*]] = insertelement <8 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
+; MAXBW-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; MAXBW: [[VEC_EPILOG_VECTOR_BODY]]:
+; MAXBW-NEXT: [[INDEX5:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT9:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; MAXBW-NEXT: [[VEC_PHI6:%.*]] = phi <8 x i32> [ [[TMP17]], %[[VEC_EPILOG_PH]] ], [ [[TMP14:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; MAXBW-NEXT: [[TMP9:%.*]] = getelementptr inbounds i8, ptr [[WEIGHTS]], i64 [[INDEX5]]
+; MAXBW-NEXT: [[WIDE_LOAD7:%.*]] = load <8 x i8>, ptr [[TMP9]], align 1
+; MAXBW-NEXT: [[TMP10:%.*]] = getelementptr inbounds i8, ptr [[INPUT]], i64 [[INDEX5]]
+; MAXBW-NEXT: [[WIDE_LOAD8:%.*]] = load <8 x i8>, ptr [[TMP10]], align 1
+; MAXBW-NEXT: [[TMP11:%.*]] = sext <8 x i8> [[WIDE_LOAD7]] to <8 x i32>
+; MAXBW-NEXT: [[TMP12:%.*]] = zext <8 x i8> [[WIDE_LOAD8]] to <8 x i32>
+; MAXBW-NEXT: [[TMP13:%.*]] = mul nsw <8 x i32> [[TMP11]], [[TMP12]]
+; MAXBW-NEXT: [[TMP14]] = add <8 x i32> [[TMP13]], [[VEC_PHI6]]
+; MAXBW-NEXT: [[INDEX_NEXT9]] = add nuw i64 [[INDEX5]], 8
+; MAXBW-NEXT: [[TMP15:%.*]] = icmp eq i64 [[INDEX_NEXT9]], [[N_VEC4]]
+; MAXBW-NEXT: br i1 [[TMP15]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; MAXBW: [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; MAXBW-NEXT: [[TMP16:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP14]])
+; MAXBW-NEXT: [[CMP_N10:%.*]] = icmp eq i64 [[N]], [[N_VEC4]]
+; MAXBW-NEXT: br i1 [[CMP_N10]], label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; MAXBW: [[VEC_EPILOG_SCALAR_PH]]:
+; MAXBW-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC4]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; MAXBW-NEXT: [[BC_MERGE_RDX10:%.*]] = phi i32 [ [[TMP16]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP8]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; MAXBW-NEXT: br label %[[LOOP:.*]]
+; MAXBW: [[LOOP]]:
+; MAXBW-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; MAXBW-NEXT: [[SUM:%.*]] = phi i32 [ [[BC_MERGE_RDX10]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[ADD:%.*]], %[[LOOP]] ]
+; MAXBW-NEXT: [[GW:%.*]] = getelementptr inbounds i8, ptr [[WEIGHTS]], i64 [[IV]]
+; MAXBW-NEXT: [[W:%.*]] = load i8, ptr [[GW]], align 1
+; MAXBW-NEXT: [[GI:%.*]] = getelementptr inbounds i8, ptr [[INPUT]], i64 [[IV]]
+; MAXBW-NEXT: [[I:%.*]] = load i8, ptr [[GI]], align 1
+; MAXBW-NEXT: [[WE:%.*]] = sext i8 [[W]] to i32
+; MAXBW-NEXT: [[IE:%.*]] = zext i8 [[I]] to i32
+; MAXBW-NEXT: [[MUL:%.*]] = mul nsw i32 [[WE]], [[IE]]
+; MAXBW-NEXT: [[ADD]] = add nsw i32 [[MUL]], [[SUM]]
+; MAXBW-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; MAXBW-NEXT: [[COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; MAXBW-NEXT: br i1 [[COND]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; MAXBW: [[EXIT]]:
+; MAXBW-NEXT: [[ADD_LCSSA:%.*]] = phi i32 [ [[ADD]], %[[LOOP]] ], [ [[TMP8]], %[[MIDDLE_BLOCK]] ], [ [[TMP16]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; MAXBW-NEXT: store i32 [[ADD_LCSSA]], ptr [[OUTPUT]], align 4
+; MAXBW-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %sum = phi i32 [ 0, %entry ], [ %add, %loop ]
+ %gw = getelementptr inbounds i8, ptr %weights, i64 %iv
+ %w = load i8, ptr %gw
+ %gi = getelementptr inbounds i8, ptr %input, i64 %iv
+ %i = load i8, ptr %gi
+ %we = sext i8 %w to i32
+ %ie = zext i8 %i to i32
+ %mul = mul nsw i32 %we, %ie
+ %add = add nsw i32 %mul, %sum
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cond = icmp eq i64 %iv.next, %n
+ br i1 %cond, label %exit, label %loop
+
+exit:
+ store i32 %add, ptr %output
+ ret void
+}
+
+; Four i8 loads, sext to i32, multiply pairs, add, store i32.
+; Smallest type=i8, widest=i32. MaxBW widens to VF=64.
+define void @multi_sext_dot(ptr noalias %a, ptr noalias %b,
+; MAXBW-LABEL: define void @multi_sext_dot(
+; MAXBW-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]], ptr noalias [[D:%.*]], ptr noalias [[OUT:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; MAXBW-NEXT: [[ITER_CHECK:.*]]:
+; MAXBW-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; MAXBW-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; MAXBW: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; MAXBW-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 64
+; MAXBW-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; MAXBW: [[VECTOR_PH]]:
+; MAXBW-NEXT: [[N_MOD_VF:%.*]] = and i64 [[N]], 63
+; MAXBW-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; MAXBW-NEXT: br label %[[VECTOR_BODY:.*]]
+; MAXBW: [[VECTOR_BODY]]:
+; MAXBW-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; MAXBW-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD:%.*]] = load <64 x i8>, ptr [[TMP0]], align 1
+; MAXBW-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD2:%.*]] = load <64 x i8>, ptr [[TMP1]], align 1
+; MAXBW-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[C]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD3:%.*]] = load <64 x i8>, ptr [[TMP2]], align 1
+; MAXBW-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[D]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD4:%.*]] = load <64 x i8>, ptr [[TMP3]], align 1
+; MAXBW-NEXT: [[TMP4:%.*]] = sext <64 x i8> [[WIDE_LOAD]] to <64 x i32>
+; MAXBW-NEXT: [[TMP5:%.*]] = sext <64 x i8> [[WIDE_LOAD2]] to <64 x i32>
+; MAXBW-NEXT: [[TMP6:%.*]] = sext <64 x i8> [[WIDE_LOAD3]] to <64 x i32>
+; MAXBW-NEXT: [[TMP7:%.*]] = sext <64 x i8> [[WIDE_LOAD4]] to <64 x i32>
+; MAXBW-NEXT: [[TMP8:%.*]] = mul nsw <64 x i32> [[TMP4]], [[TMP5]]
+; MAXBW-NEXT: [[TMP9:%.*]] = mul nsw <64 x i32> [[TMP6]], [[TMP7]]
+; MAXBW-NEXT: [[TMP10:%.*]] = add nsw <64 x i32> [[TMP8]], [[TMP9]]
+; MAXBW-NEXT: [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[OUT]], i64 [[INDEX]]
+; MAXBW-NEXT: store <64 x i32> [[TMP10]], ptr [[TMP11]], align 4
+; MAXBW-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
+; MAXBW-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; MAXBW-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; MAXBW: [[MIDDLE_BLOCK]]:
+; MAXBW-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; MAXBW-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; MAXBW: [[VEC_EPILOG_ITER_CHECK]]:
+; MAXBW-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
+; MAXBW-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
+; MAXBW: [[VEC_EPILOG_PH]]:
+; MAXBW-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; MAXBW-NEXT: [[N_MOD_VF5:%.*]] = and i64 [[N]], 7
+; MAXBW-NEXT: [[N_VEC6:%.*]] = sub i64 [[N]], [[N_MOD_VF5]]
+; MAXBW-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; MAXBW: [[VEC_EPILOG_VECTOR_BODY]]:
+; MAXBW-NEXT: [[INDEX7:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT12:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; MAXBW-NEXT: [[TMP13:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX7]]
+; MAXBW-NEXT: [[WIDE_LOAD8:%.*]] = load <8 x i8>, ptr [[TMP13]], align 1
+; MAXBW-NEXT: [[TMP14:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[INDEX7]]
+; MAXBW-NEXT: [[WIDE_LOAD9:%.*]] = load <8 x i8>, ptr [[TMP14]], align 1
+; MAXBW-NEXT: [[TMP15:%.*]] = getelementptr inbounds i8, ptr [[C]], i64 [[INDEX7]]
+; MAXBW-NEXT: [[WIDE_LOAD10:%.*]] = load <8 x i8>, ptr [[TMP15]], align 1
+; MAXBW-NEXT: [[TMP16:%.*]] = getelementptr inbounds i8, ptr [[D]], i64 [[INDEX7]]
+; MAXBW-NEXT: [[WIDE_LOAD11:%.*]] = load <8 x i8>, ptr [[TMP16]], align 1
+; MAXBW-NEXT: [[TMP17:%.*]] = sext <8 x i8> [[WIDE_LOAD8]] to <8 x i32>
+; MAXBW-NEXT: [[TMP18:%.*]] = sext <8 x i8> [[WIDE_LOAD9]] to <8 x i32>
+; MAXBW-NEXT: [[TMP19:%.*]] = sext <8 x i8> [[WIDE_LOAD10]] to <8 x i32>
+; MAXBW-NEXT: [[TMP20:%.*]] = sext <8 x i8> [[WIDE_LOAD11]] to <8 x i32>
+; MAXBW-NEXT: [[TMP21:%.*]] = mul nsw <8 x i32> [[TMP17]], [[TMP18]]
+; MAXBW-NEXT: [[TMP22:%.*]] = mul nsw <8 x i32> [[TMP19]], [[TMP20]]
+; MAXBW-NEXT: [[TMP23:%.*]] = add nsw <8 x i32> [[TMP21]], [[TMP22]]
+; MAXBW-NEXT: [[TMP24:%.*]] = getelementptr inbounds i32, ptr [[OUT]], i64 [[INDEX7]]
+; MAXBW-NEXT: store <8 x i32> [[TMP23]], ptr [[TMP24]], align 4
+; MAXBW-NEXT: [[INDEX_NEXT12]] = add nuw i64 [[INDEX7]], 8
+; MAXBW-NEXT: [[TMP25:%.*]] = icmp eq i64 [[INDEX_NEXT12]], [[N_VEC6]]
+; MAXBW-NEXT: br i1 [[TMP25]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; MAXBW: [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; MAXBW-NEXT: [[CMP_N13:%.*]] = icmp eq i64 [[N]], [[N_VEC6]]
+; MAXBW-NEXT: br i1 [[CMP_N13]], label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; MAXBW: [[VEC_EPILOG_SCALAR_PH]]:
+; MAXBW-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC6]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; MAXBW-NEXT: br label %[[LOOP:.*]]
+; MAXBW: [[LOOP]]:
+; MAXBW-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; MAXBW-NEXT: [[GA:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV]]
+; MAXBW-NEXT: [[LA:%.*]] = load i8, ptr [[GA]], align 1
+; MAXBW-NEXT: [[GB:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV]]
+; MAXBW-NEXT: [[LB:%.*]] = load i8, ptr [[GB]], align 1
+; MAXBW-NEXT: [[GC:%.*]] = getelementptr inbounds i8, ptr [[C]], i64 [[IV]]
+; MAXBW-NEXT: [[LC:%.*]] = load i8, ptr [[GC]], align 1
+; MAXBW-NEXT: [[GD:%.*]] = getelementptr inbounds i8, ptr [[D]], i64 [[IV]]
+; MAXBW-NEXT: [[LD:%.*]] = load i8, ptr [[GD]], align 1
+; MAXBW-NEXT: [[EA:%.*]] = sext i8 [[LA]] to i32
+; MAXBW-NEXT: [[EB:%.*]] = sext i8 [[LB]] to i32
+; MAXBW-NEXT: [[EC:%.*]] = sext i8 [[LC]] to i32
+; MAXBW-NEXT: [[ED:%.*]] = sext i8 [[LD]] to i32
+; MAXBW-NEXT: [[M1:%.*]] = mul nsw i32 [[EA]], [[EB]]
+; MAXBW-NEXT: [[M2:%.*]] = mul nsw i32 [[EC]], [[ED]]
+; MAXBW-NEXT: [[SUM:%.*]] = add nsw i32 [[M1]], [[M2]]
+; MAXBW-NEXT: [[GO:%.*]] = getelementptr inbounds i32, ptr [[OUT]], i64 [[IV]]
+; MAXBW-NEXT: store i32 [[SUM]], ptr [[GO]], align 4
+; MAXBW-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; MAXBW-NEXT: [[COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; MAXBW-NEXT: br i1 [[COND]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
+; MAXBW: [[EXIT]]:
+; MAXBW-NEXT: ret void
+;
+ ptr noalias %c, ptr noalias %d,
+ ptr noalias %out, i64 %n) {
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %ga = getelementptr inbounds i8, ptr %a, i64 %iv
+ %la = load i8, ptr %ga
+ %gb = getelementptr inbounds i8, ptr %b, i64 %iv
+ %lb = load i8, ptr %gb
+ %gc = getelementptr inbounds i8, ptr %c, i64 %iv
+ %lc = load i8, ptr %gc
+ %gd = getelementptr inbounds i8, ptr %d, i64 %iv
+ %ld = load i8, ptr %gd
+ %ea = sext i8 %la to i32
+ %eb = sext i8 %lb to i32
+ %ec = sext i8 %lc to i32
+ %ed = sext i8 %ld to i32
+ %m1 = mul nsw i32 %ea, %eb
+ %m2 = mul nsw i32 %ec, %ed
+ %sum = add nsw i32 %m1, %m2
+ %go = getelementptr inbounds i32, ptr %out, i64 %iv
+ store i32 %sum, ptr %go
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cond = icmp eq i64 %iv.next, %n
+ br i1 %cond, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+; i16 add — smallest = widest = i16. Both MaxBW and default select
+; VF = 512/16 = 32. Control case verifying no regression.
+define void @add_i16(ptr noalias %a, ptr noalias %b, ptr noalias %c, i64 %n) {
+; MAXBW-LABEL: define void @add_i16(
+; MAXBW-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; MAXBW-NEXT: [[ITER_CHECK:.*]]:
+; MAXBW-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; MAXBW-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; MAXBW: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; MAXBW-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; MAXBW-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; MAXBW: [[VECTOR_PH]]:
+; MAXBW-NEXT: [[N_MOD_VF:%.*]] = and i64 [[N]], 31
+; MAXBW-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; MAXBW-NEXT: br label %[[VECTOR_BODY:.*]]
+; MAXBW: [[VECTOR_BODY]]:
+; MAXBW-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; MAXBW-NEXT: [[TMP0:%.*]] = getelementptr inbounds i16, ptr [[A]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i16>, ptr [[TMP0]], align 2
+; MAXBW-NEXT: [[TMP1:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD2:%.*]] = load <32 x i16>, ptr [[TMP1]], align 2
+; MAXBW-NEXT: [[TMP2:%.*]] = add <32 x i16> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; MAXBW-NEXT: [[TMP3:%.*]] = getelementptr inbounds i16, ptr [[C]], i64 [[INDEX]]
+; MAXBW-NEXT: store <32 x i16> [[TMP2]], ptr [[TMP3]], align 2
+; MAXBW-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; MAXBW-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; MAXBW-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
+; MAXBW: [[MIDDLE_BLOCK]]:
+; MAXBW-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; MAXBW-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; MAXBW: [[VEC_EPILOG_ITER_CHECK]]:
+; MAXBW-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
+; MAXBW-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF10:![0-9]+]]
+; MAXBW: [[VEC_EPILOG_PH]]:
+; MAXBW-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; MAXBW-NEXT: [[N_MOD_VF3:%.*]] = and i64 [[N]], 3
+; MAXBW-NEXT: [[N_VEC4:%.*]] = sub i64 [[N]], [[N_MOD_VF3]]
+; MAXBW-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; MAXBW: [[VEC_EPILOG_VECTOR_BODY]]:
+; MAXBW-NEXT: [[INDEX5:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT8:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; MAXBW-NEXT: [[TMP5:%.*]] = getelementptr inbounds i16, ptr [[A]], i64 [[INDEX5]]
+; MAXBW-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i16>, ptr [[TMP5]], align 2
+; MAXBW-NEXT: [[TMP6:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[INDEX5]]
+; MAXBW-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i16>, ptr [[TMP6]], align 2
+; MAXBW-NEXT: [[TMP7:%.*]] = add <4 x i16> [[WIDE_LOAD6]], [[WIDE_LOAD7]]
+; MAXBW-NEXT: [[TMP8:%.*]] = getelementptr inbounds i16, ptr [[C]], i64 [[INDEX5]]
+; MAXBW-NEXT: store <4 x i16> [[TMP7]], ptr [[TMP8]], align 2
+; MAXBW-NEXT: [[INDEX_NEXT8]] = add nuw i64 [[INDEX5]], 4
+; MAXBW-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT8]], [[N_VEC4]]
+; MAXBW-NEXT: br i1 [[TMP9]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; MAXBW: [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; MAXBW-NEXT: [[CMP_N9:%.*]] = icmp eq i64 [[N]], [[N_VEC4]]
+; MAXBW-NEXT: br i1 [[CMP_N9]], label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; MAXBW: [[VEC_EPILOG_SCALAR_PH]]:
+; MAXBW-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC4]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; MAXBW-NEXT: br label %[[LOOP:.*]]
+; MAXBW: [[LOOP]]:
+; MAXBW-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; MAXBW-NEXT: [[GA:%.*]] = getelementptr inbounds i16, ptr [[A]], i64 [[IV]]
+; MAXBW-NEXT: [[LA:%.*]] = load i16, ptr [[GA]], align 2
+; MAXBW-NEXT: [[GB:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[IV]]
+; MAXBW-NEXT: [[LB:%.*]] = load i16, ptr [[GB]], align 2
+; MAXBW-NEXT: [[ADD:%.*]] = add i16 [[LA]], [[LB]]
+; MAXBW-NEXT: [[GC:%.*]] = getelementptr inbounds i16, ptr [[C]], i64 [[IV]]
+; MAXBW-NEXT: store i16 [[ADD]], ptr [[GC]], align 2
+; MAXBW-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; MAXBW-NEXT: [[COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; MAXBW-NEXT: br i1 [[COND]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP12:![0-9]+]]
+; MAXBW: [[EXIT]]:
+; MAXBW-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %ga = getelementptr inbounds i16, ptr %a, i64 %iv
+ %la = load i16, ptr %ga
+ %gb = getelementptr inbounds i16, ptr %b, i64 %iv
+ %lb = load i16, ptr %gb
+ %add = add i16 %la, %lb
+ %gc = getelementptr inbounds i16, ptr %c, i64 %iv
+ store i16 %add, ptr %gc
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cond = icmp eq i64 %iv.next, %n
+ br i1 %cond, label %exit, label %loop
+
+exit:
+ ret void
+}
+;.
+; MAXBW: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; MAXBW: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; MAXBW: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; MAXBW: [[PROF3]] = !{!"branch_weights", i32 8, i32 56}
+; MAXBW: [[LOOP4]] = distinct !{[[LOOP4]], [[META1]], [[META2]]}
+; MAXBW: [[LOOP5]] = distinct !{[[LOOP5]], [[META2]], [[META1]]}
+; MAXBW: [[LOOP6]] = distinct !{[[LOOP6]], [[META1]], [[META2]]}
+; MAXBW: [[LOOP7]] = distinct !{[[LOOP7]], [[META1]], [[META2]]}
+; MAXBW: [[LOOP8]] = distinct !{[[LOOP8]], [[META2]], [[META1]]}
+; MAXBW: [[LOOP9]] = distinct !{[[LOOP9]], [[META1]], [[META2]]}
+; MAXBW: [[PROF10]] = !{!"branch_weights", i32 4, i32 28}
+; MAXBW: [[LOOP11]] = distinct !{[[LOOP11]], [[META1]], [[META2]]}
+; MAXBW: [[LOOP12]] = distinct !{[[LOOP12]], [[META2]], [[META1]]}
+;.
diff --git a/llvm/test/Transforms/LoopVectorize/X86/no_fpmath.ll b/llvm/test/Transforms/LoopVectorize/X86/no_fpmath.ll
index ae77b4270ab3e..2056547c2c988 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/no_fpmath.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/no_fpmath.ll
@@ -2,7 +2,7 @@
; CHECK: remark: no_fpmath.c:6:11: loop not vectorized: cannot prove it is safe to reorder floating-point operations
; CHECK: remark: no_fpmath.c:6:14: loop not vectorized
-; CHECK: remark: no_fpmath.c:17:14: vectorized loop (vectorization width: 2, interleaved count: 2)
+; CHECK: remark: no_fpmath.c:17:14: vectorized loop (vectorization width: 4, interleaved count: 2)
target datalayout = "e-m:o-i64:64-f80:128-n8:16:32:64-S128"
target triple = "x86_64-apple-macosx10.10.0"
diff --git a/llvm/test/Transforms/LoopVectorize/X86/no_fpmath_with_hotness.ll b/llvm/test/Transforms/LoopVectorize/X86/no_fpmath_with_hotness.ll
index cbedbf7fd10c0..6406756f1eead 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/no_fpmath_with_hotness.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/no_fpmath_with_hotness.ll
@@ -2,7 +2,7 @@
; CHECK: remark: no_fpmath.c:6:11: loop not vectorized: cannot prove it is safe to reorder floating-point operations (hotness: 300)
; CHECK: remark: no_fpmath.c:6:14: loop not vectorized
-; CHECK: remark: no_fpmath.c:17:14: vectorized loop (vectorization width: 2, interleaved count: 1) (hotness: 300)
+; CHECK: remark: no_fpmath.c:17:14: vectorized loop (vectorization width: 4, interleaved count: 1) (hotness: 300)
target datalayout = "e-m:o-i64:64-f80:128-n8:16:32:64-S128"
target triple = "x86_64-apple-macosx10.10.0"
diff --git a/llvm/test/Transforms/LoopVectorize/X86/nondetermisitic-widening-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/nondetermisitic-widening-cost.ll
index 9e473b373faa8..dbd183e18b13d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/nondetermisitic-widening-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/nondetermisitic-widening-cost.ll
@@ -15,47 +15,75 @@ define float @fun(i64 %0, float %1, ptr noalias %a, ptr noalias %b, i64 %len) #
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[TMP3:%.*]] = shl i64 [[TMP0]], 2
; CHECK-NEXT: [[TMP5:%.*]] = sub i64 [[TMP3]], 1
-; CHECK-NEXT: [[DIFF_CHECK:%.*]] = icmp ult i64 [[TMP5]], 15
+; CHECK-NEXT: [[DIFF_CHECK:%.*]] = icmp ult i64 [[TMP5]], 31
; CHECK-NEXT: br i1 [[DIFF_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr [4 x i8], ptr [[VLA]], i64 [[TMP0]]
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x float> poison, float [[TMP1]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x float> [[BROADCAST_SPLATINSERT]], <4 x float> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x float> poison, float [[TMP1]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x float> [[BROADCAST_SPLATINSERT]], <8 x float> poison, <8 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP6:%.*]] = add i64 [[INDEX]], 1
; CHECK-NEXT: [[TMP7:%.*]] = add i64 [[INDEX]], 2
; CHECK-NEXT: [[TMP8:%.*]] = add i64 [[INDEX]], 3
+; CHECK-NEXT: [[TMP12:%.*]] = add i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[INDEX]], 5
+; CHECK-NEXT: [[TMP10:%.*]] = add i64 [[INDEX]], 6
+; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[INDEX]], 7
; CHECK-NEXT: [[TMP15:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[INDEX]]
; CHECK-NEXT: [[TMP16:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP6]]
; CHECK-NEXT: [[TMP17:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP7]]
; CHECK-NEXT: [[TMP18:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP8]]
+; CHECK-NEXT: [[TMP20:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP12]]
+; CHECK-NEXT: [[TMP21:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP9]]
+; CHECK-NEXT: [[TMP22:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP10]]
+; CHECK-NEXT: [[TMP19:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP11]]
; CHECK-NEXT: [[TMP23:%.*]] = load ptr, ptr [[TMP15]], align 8
; CHECK-NEXT: [[TMP24:%.*]] = load ptr, ptr [[TMP16]], align 8
; CHECK-NEXT: [[TMP25:%.*]] = load ptr, ptr [[TMP17]], align 8
; CHECK-NEXT: [[TMP26:%.*]] = load ptr, ptr [[TMP18]], align 8
+; CHECK-NEXT: [[TMP28:%.*]] = load ptr, ptr [[TMP20]], align 8
+; CHECK-NEXT: [[TMP29:%.*]] = load ptr, ptr [[TMP21]], align 8
+; CHECK-NEXT: [[TMP30:%.*]] = load ptr, ptr [[TMP22]], align 8
+; CHECK-NEXT: [[TMP27:%.*]] = load ptr, ptr [[TMP19]], align 8
; CHECK-NEXT: [[TMP31:%.*]] = load i64, ptr [[TMP23]], align 8
; CHECK-NEXT: [[TMP32:%.*]] = load i64, ptr [[TMP24]], align 8
; CHECK-NEXT: [[TMP36:%.*]] = load i64, ptr [[TMP25]], align 8
; CHECK-NEXT: [[TMP34:%.*]] = load i64, ptr [[TMP26]], align 8
+; CHECK-NEXT: [[TMP60:%.*]] = load i64, ptr [[TMP28]], align 8
+; CHECK-NEXT: [[TMP33:%.*]] = load i64, ptr [[TMP29]], align 8
+; CHECK-NEXT: [[TMP62:%.*]] = load i64, ptr [[TMP30]], align 8
+; CHECK-NEXT: [[TMP35:%.*]] = load i64, ptr [[TMP27]], align 8
; CHECK-NEXT: [[TMP44:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP31]]
; CHECK-NEXT: [[TMP45:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP32]]
; CHECK-NEXT: [[TMP46:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP36]]
; CHECK-NEXT: [[TMP47:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP34]]
+; CHECK-NEXT: [[TMP64:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP60]]
+; CHECK-NEXT: [[TMP65:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP33]]
+; CHECK-NEXT: [[TMP66:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP62]]
+; CHECK-NEXT: [[TMP43:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP35]]
; CHECK-NEXT: [[TMP71:%.*]] = load float, ptr [[TMP44]], align 4
; CHECK-NEXT: [[TMP72:%.*]] = load float, ptr [[TMP45]], align 4
; CHECK-NEXT: [[TMP73:%.*]] = load float, ptr [[TMP46]], align 4
; CHECK-NEXT: [[TMP74:%.*]] = load float, ptr [[TMP47]], align 4
-; CHECK-NEXT: [[TMP75:%.*]] = insertelement <4 x float> poison, float [[TMP71]], i64 0
-; CHECK-NEXT: [[TMP76:%.*]] = insertelement <4 x float> [[TMP75]], float [[TMP72]], i64 1
-; CHECK-NEXT: [[TMP77:%.*]] = insertelement <4 x float> [[TMP76]], float [[TMP73]], i64 2
-; CHECK-NEXT: [[TMP78:%.*]] = insertelement <4 x float> [[TMP77]], float [[TMP74]], i64 3
+; CHECK-NEXT: [[TMP48:%.*]] = load float, ptr [[TMP64]], align 4
+; CHECK-NEXT: [[TMP49:%.*]] = load float, ptr [[TMP65]], align 4
+; CHECK-NEXT: [[TMP50:%.*]] = load float, ptr [[TMP66]], align 4
+; CHECK-NEXT: [[TMP51:%.*]] = load float, ptr [[TMP43]], align 4
+; CHECK-NEXT: [[TMP52:%.*]] = insertelement <8 x float> poison, float [[TMP71]], i64 0
+; CHECK-NEXT: [[TMP53:%.*]] = insertelement <8 x float> [[TMP52]], float [[TMP72]], i64 1
+; CHECK-NEXT: [[TMP54:%.*]] = insertelement <8 x float> [[TMP53]], float [[TMP73]], i64 2
+; CHECK-NEXT: [[TMP55:%.*]] = insertelement <8 x float> [[TMP54]], float [[TMP74]], i64 3
+; CHECK-NEXT: [[TMP56:%.*]] = insertelement <8 x float> [[TMP55]], float [[TMP48]], i64 4
+; CHECK-NEXT: [[TMP57:%.*]] = insertelement <8 x float> [[TMP56]], float [[TMP49]], i64 5
+; CHECK-NEXT: [[TMP58:%.*]] = insertelement <8 x float> [[TMP57]], float [[TMP50]], i64 6
+; CHECK-NEXT: [[TMP59:%.*]] = insertelement <8 x float> [[TMP58]], float [[TMP51]], i64 7
; CHECK-NEXT: [[TMP61:%.*]] = getelementptr [4 x i8], ptr [[VLA]], i64 [[INDEX]]
-; CHECK-NEXT: store <4 x float> [[TMP78]], ptr [[TMP61]], align 4
+; CHECK-NEXT: store <8 x float> [[TMP59]], ptr [[TMP61]], align 4
; CHECK-NEXT: [[TMP63:%.*]] = getelementptr [4 x i8], ptr [[TMP4]], i64 [[INDEX]]
-; CHECK-NEXT: store <4 x float> [[BROADCAST_SPLAT]], ptr [[TMP63]], align 4
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: store <8 x float> [[BROADCAST_SPLAT]], ptr [[TMP63]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; CHECK-NEXT: [[TMP37:%.*]] = icmp eq i64 [[INDEX_NEXT]], 128
; CHECK-NEXT: br i1 [[TMP37]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/pr47437.ll b/llvm/test/Transforms/LoopVectorize/X86/pr47437.ll
index e546743472b71..533e35d10d73a 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/pr47437.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/pr47437.ll
@@ -333,45 +333,82 @@ define void @test_muladd(ptr noalias nocapture %d1, ptr noalias nocapture readon
; AVX2-NEXT: entry:
; AVX2-NEXT: [[CMP30:%.*]] = icmp sgt i32 [[N:%.*]], 0
; AVX2-NEXT: br i1 [[CMP30]], label [[FOR_BODY_PREHEADER:%.*]], label [[FOR_END:%.*]]
-; AVX2: for.body.preheader:
+; AVX2: iter.check:
; AVX2-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[N]] to i64
-; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
+; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_PH:%.*]]
+; AVX2: vector.main.loop.iter.check:
+; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
+; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK1]], label [[VEC_EPILOG_PH:%.*]], label [[VECTOR_PH1:%.*]]
; AVX2: vector.ph:
-; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
+; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
; AVX2-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX2-NEXT: br label [[VECTOR_BODY:%.*]]
; AVX2: vector.body:
-; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH1]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP1:%.*]] = shl nuw nsw i64 [[INDEX]], 1
; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i16, ptr [[S1:%.*]], i64 [[TMP1]]
-; AVX2-NEXT: [[WIDE_VEC:%.*]] = load <16 x i16>, ptr [[TMP2]], align 2
-; AVX2-NEXT: [[STRIDED_VEC:%.*]] = shufflevector <16 x i16> [[WIDE_VEC]], <16 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
-; AVX2-NEXT: [[STRIDED_VEC1:%.*]] = shufflevector <16 x i16> [[WIDE_VEC]], <16 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
-; AVX2-NEXT: [[TMP4:%.*]] = sext <8 x i16> [[STRIDED_VEC]] to <8 x i32>
+; AVX2-NEXT: [[WIDE_VEC:%.*]] = load <32 x i16>, ptr [[TMP2]], align 2
+; AVX2-NEXT: [[STRIDED_VEC:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30>
+; AVX2-NEXT: [[STRIDED_VEC2:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <16 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
+; AVX2-NEXT: [[TMP3:%.*]] = sext <16 x i16> [[STRIDED_VEC]] to <16 x i32>
; AVX2-NEXT: [[TMP5:%.*]] = getelementptr inbounds i16, ptr [[S2:%.*]], i64 [[TMP1]]
-; AVX2-NEXT: [[WIDE_VEC2:%.*]] = load <16 x i16>, ptr [[TMP5]], align 2
-; AVX2-NEXT: [[STRIDED_VEC3:%.*]] = shufflevector <16 x i16> [[WIDE_VEC2]], <16 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
-; AVX2-NEXT: [[STRIDED_VEC4:%.*]] = shufflevector <16 x i16> [[WIDE_VEC2]], <16 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
-; AVX2-NEXT: [[TMP7:%.*]] = sext <8 x i16> [[STRIDED_VEC3]] to <8 x i32>
-; AVX2-NEXT: [[TMP8:%.*]] = mul nsw <8 x i32> [[TMP7]], [[TMP4]]
-; AVX2-NEXT: [[TMP9:%.*]] = sext <8 x i16> [[STRIDED_VEC1]] to <8 x i32>
-; AVX2-NEXT: [[TMP10:%.*]] = sext <8 x i16> [[STRIDED_VEC4]] to <8 x i32>
-; AVX2-NEXT: [[TMP11:%.*]] = mul nsw <8 x i32> [[TMP10]], [[TMP9]]
-; AVX2-NEXT: [[TMP12:%.*]] = add nsw <8 x i32> [[TMP11]], [[TMP8]]
+; AVX2-NEXT: [[WIDE_VEC3:%.*]] = load <32 x i16>, ptr [[TMP5]], align 2
+; AVX2-NEXT: [[STRIDED_VEC4:%.*]] = shufflevector <32 x i16> [[WIDE_VEC3]], <32 x i16> poison, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30>
+; AVX2-NEXT: [[STRIDED_VEC5:%.*]] = shufflevector <32 x i16> [[WIDE_VEC3]], <32 x i16> poison, <16 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
+; AVX2-NEXT: [[TMP11:%.*]] = sext <16 x i16> [[STRIDED_VEC4]] to <16 x i32>
+; AVX2-NEXT: [[TMP6:%.*]] = mul nsw <16 x i32> [[TMP11]], [[TMP3]]
+; AVX2-NEXT: [[TMP7:%.*]] = sext <16 x i16> [[STRIDED_VEC2]] to <16 x i32>
+; AVX2-NEXT: [[TMP8:%.*]] = sext <16 x i16> [[STRIDED_VEC5]] to <16 x i32>
+; AVX2-NEXT: [[TMP9:%.*]] = mul nsw <16 x i32> [[TMP8]], [[TMP7]]
+; AVX2-NEXT: [[TMP10:%.*]] = add nsw <16 x i32> [[TMP9]], [[TMP6]]
; AVX2-NEXT: [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[D1:%.*]], i64 [[INDEX]]
-; AVX2-NEXT: store <8 x i32> [[TMP12]], ptr [[TMP13]], align 4
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; AVX2-NEXT: store <16 x i32> [[TMP10]], ptr [[TMP13]], align 4
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX2-NEXT: [[TMP15:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX2-NEXT: br i1 [[TMP15]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; AVX2: middle.block:
; AVX2-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
-; AVX2-NEXT: br i1 [[CMP_N]], label [[FOR_END_LOOPEXIT:%.*]], label [[SCALAR_PH]]
-; AVX2: scalar.ph:
-; AVX2-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[MIDDLE_BLOCK]] ], [ 0, [[FOR_BODY_PREHEADER]] ]
+; AVX2-NEXT: br i1 [[CMP_N]], label [[FOR_END_LOOPEXIT:%.*]], label [[VEC_EPILOG_ITER_CHECK:%.*]]
+; AVX2: vec.epilog.iter.check:
+; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
+; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label [[SCALAR_PH]], label [[VEC_EPILOG_PH]], !prof [[PROF3:![0-9]+]]
+; AVX2: vec.epilog.ph:
+; AVX2-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[VECTOR_PH]] ]
+; AVX2-NEXT: [[TMP26:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
+; AVX2-NEXT: [[N_VEC6:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[TMP26]]
; AVX2-NEXT: br label [[FOR_BODY:%.*]]
+; AVX2: vec.epilog.vector.body:
+; AVX2-NEXT: [[INDEX7:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], [[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT14:%.*]], [[FOR_BODY]] ]
+; AVX2-NEXT: [[TMP14:%.*]] = shl nuw nsw i64 [[INDEX7]], 1
+; AVX2-NEXT: [[TMP27:%.*]] = getelementptr inbounds i16, ptr [[S1]], i64 [[TMP14]]
+; AVX2-NEXT: [[WIDE_VEC8:%.*]] = load <8 x i16>, ptr [[TMP27]], align 2
+; AVX2-NEXT: [[STRIDED_VEC9:%.*]] = shufflevector <8 x i16> [[WIDE_VEC8]], <8 x i16> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; AVX2-NEXT: [[STRIDED_VEC10:%.*]] = shufflevector <8 x i16> [[WIDE_VEC8]], <8 x i16> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; AVX2-NEXT: [[TMP28:%.*]] = sext <4 x i16> [[STRIDED_VEC9]] to <4 x i32>
+; AVX2-NEXT: [[TMP29:%.*]] = getelementptr inbounds i16, ptr [[S2]], i64 [[TMP14]]
+; AVX2-NEXT: [[WIDE_VEC11:%.*]] = load <8 x i16>, ptr [[TMP29]], align 2
+; AVX2-NEXT: [[STRIDED_VEC12:%.*]] = shufflevector <8 x i16> [[WIDE_VEC11]], <8 x i16> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; AVX2-NEXT: [[STRIDED_VEC13:%.*]] = shufflevector <8 x i16> [[WIDE_VEC11]], <8 x i16> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; AVX2-NEXT: [[TMP30:%.*]] = sext <4 x i16> [[STRIDED_VEC12]] to <4 x i32>
+; AVX2-NEXT: [[TMP31:%.*]] = mul nsw <4 x i32> [[TMP30]], [[TMP28]]
+; AVX2-NEXT: [[TMP32:%.*]] = sext <4 x i16> [[STRIDED_VEC10]] to <4 x i32>
+; AVX2-NEXT: [[TMP33:%.*]] = sext <4 x i16> [[STRIDED_VEC13]] to <4 x i32>
+; AVX2-NEXT: [[TMP22:%.*]] = mul nsw <4 x i32> [[TMP33]], [[TMP32]]
+; AVX2-NEXT: [[TMP23:%.*]] = add nsw <4 x i32> [[TMP22]], [[TMP31]]
+; AVX2-NEXT: [[TMP24:%.*]] = getelementptr inbounds i32, ptr [[D1]], i64 [[INDEX7]]
+; AVX2-NEXT: store <4 x i32> [[TMP23]], ptr [[TMP24]], align 4
+; AVX2-NEXT: [[INDEX_NEXT14]] = add nuw i64 [[INDEX7]], 4
+; AVX2-NEXT: [[TMP25:%.*]] = icmp eq i64 [[INDEX_NEXT14]], [[N_VEC6]]
+; AVX2-NEXT: br i1 [[TMP25]], label [[VEC_EPILOG_MIDDLE_BLOCK:%.*]], label [[FOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; AVX2: vec.epilog.middle.block:
+; AVX2-NEXT: [[CMP_N15:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC6]]
+; AVX2-NEXT: br i1 [[CMP_N15]], label [[FOR_END_LOOPEXIT]], label [[SCALAR_PH]]
+; AVX2: vec.epilog.scalar.ph:
+; AVX2-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC6]], [[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[FOR_BODY_PREHEADER]] ]
+; AVX2-NEXT: br label [[FOR_BODY1:%.*]]
; AVX2: for.body:
-; AVX2-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], [[SCALAR_PH]] ], [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY]] ]
+; AVX2-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], [[SCALAR_PH]] ], [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY1]] ]
; AVX2-NEXT: [[TMP16:%.*]] = shl nuw nsw i64 [[INDVARS_IV]], 1
; AVX2-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds i16, ptr [[S1]], i64 [[TMP16]]
; AVX2-NEXT: [[TMP17:%.*]] = load i16, ptr [[ARRAYIDX]], align 2
@@ -393,7 +430,7 @@ define void @test_muladd(ptr noalias nocapture %d1, ptr noalias nocapture readon
; AVX2-NEXT: store i32 [[ADD18]], ptr [[ARRAYIDX20]], align 4
; AVX2-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1
; AVX2-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], [[WIDE_TRIP_COUNT]]
-; AVX2-NEXT: br i1 [[EXITCOND_NOT]], label [[FOR_END_LOOPEXIT]], label [[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; AVX2-NEXT: br i1 [[EXITCOND_NOT]], label [[FOR_END_LOOPEXIT]], label [[FOR_BODY1]], !llvm.loop [[LOOP5:![0-9]+]]
; AVX2: for.end.loopexit:
; AVX2-NEXT: br label [[FOR_END]]
; AVX2: for.end:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/reduction-crash.ll b/llvm/test/Transforms/LoopVectorize/X86/reduction-crash.ll
index 65759d545e4af..6c61d2fb9b949 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/reduction-crash.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/reduction-crash.ll
@@ -12,7 +12,7 @@ define void @pr15344(ptr noalias %ar, ptr noalias %ar2, i32 %exit.limit, i1 %con
; CHECK: [[PH]]:
; CHECK-NEXT: br i1 [[COND]], label %[[LOOP_PREHEADER:.*]], label %[[EXIT:.*]]
; CHECK: [[LOOP_PREHEADER]]:
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[EXIT_LIMIT]], 10
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[EXIT_LIMIT]], 12
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[TMP0:%.*]] = shl i32 [[EXIT_LIMIT]], 2
@@ -24,25 +24,25 @@ define void @pr15344(ptr noalias %ar, ptr noalias %ar2, i32 %exit.limit, i1 %con
; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
; CHECK-NEXT: br i1 [[FOUND_CONFLICT]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i32 [[EXIT_LIMIT]], 3
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i32 [[EXIT_LIMIT]], 7
; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[EXIT_LIMIT]], [[N_MOD_VF]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <2 x double> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP2:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI2:%.*]] = phi <2 x double> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP3:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP2]] = fadd fast <2 x double> [[VEC_PHI]], splat (double 1.000000e+00)
-; CHECK-NEXT: [[TMP3]] = fadd fast <2 x double> [[VEC_PHI2]], splat (double 1.000000e+00)
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x double> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP3:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI2:%.*]] = phi <4 x double> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP3]] = fadd fast <4 x double> [[VEC_PHI]], splat (double 1.000000e+00)
+; CHECK-NEXT: [[TMP5]] = fadd fast <4 x double> [[VEC_PHI2]], splat (double 1.000000e+00)
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds float, ptr [[AR2]], i32 [[INDEX]]
-; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds float, ptr [[TMP4]], i32 2
-; CHECK-NEXT: store <2 x float> splat (float 2.000000e+00), ptr [[TMP4]], align 4, !alias.scope [[META0:![0-9]+]], !noalias [[META3:![0-9]+]]
-; CHECK-NEXT: store <2 x float> splat (float 2.000000e+00), ptr [[TMP6]], align 4, !alias.scope [[META0]], !noalias [[META3]]
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds float, ptr [[TMP4]], i32 4
+; CHECK-NEXT: store <4 x float> splat (float 2.000000e+00), ptr [[TMP4]], align 4, !alias.scope [[META0:![0-9]+]], !noalias [[META3:![0-9]+]]
+; CHECK-NEXT: store <4 x float> splat (float 2.000000e+00), ptr [[TMP6]], align 4, !alias.scope [[META0]], !noalias [[META3]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 8
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: [[BIN_RDX:%.*]] = fadd fast <2 x double> [[TMP3]], [[TMP2]]
-; CHECK-NEXT: [[TMP8:%.*]] = call fast double @llvm.vector.reduce.fadd.v2f64(double 0.000000e+00, <2 x double> [[BIN_RDX]])
+; CHECK-NEXT: [[BIN_RDX:%.*]] = fadd fast <4 x double> [[TMP5]], [[TMP3]]
+; CHECK-NEXT: [[TMP8:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[BIN_RDX]])
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i32 [[EXIT_LIMIT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT_LOOPEXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/replicating-load-store-costs.ll b/llvm/test/Transforms/LoopVectorize/X86/replicating-load-store-costs.ll
index 35c58e0880402..8976d605a19e0 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/replicating-load-store-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/replicating-load-store-costs.ll
@@ -165,23 +165,17 @@ define void @test_store_initially_interleave(i32 %n, ptr noalias %src) #0 {
; I32-SAME: i32 [[N:%.*]], ptr noalias [[SRC:%.*]]) #[[ATTR0:[0-9]+]] {
; I32-NEXT: [[ITER_CHECK:.*:]]
; I32-NEXT: [[TMP0:%.*]] = add i32 [[N]], 1
-; I32-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ule i32 [[TMP0]], 4
-; I32-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; I32: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; I32-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ule i32 [[TMP0]], 16
+; I32-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ule i32 [[TMP0]], 8
; I32-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; I32: [[VECTOR_PH]]:
-; I32-NEXT: [[N_MOD_VF:%.*]] = and i32 [[TMP0]], 15
+; I32-NEXT: [[N_MOD_VF:%.*]] = and i32 [[TMP0]], 7
; I32-NEXT: [[TMP1:%.*]] = icmp eq i32 [[N_MOD_VF]], 0
-; I32-NEXT: [[TMP2:%.*]] = select i1 [[TMP1]], i32 16, i32 [[N_MOD_VF]]
+; I32-NEXT: [[TMP2:%.*]] = select i1 [[TMP1]], i32 8, i32 [[N_MOD_VF]]
; I32-NEXT: [[N_VEC:%.*]] = sub i32 [[TMP0]], [[TMP2]]
; I32-NEXT: br label %[[VECTOR_BODY:.*]]
; I32: [[VECTOR_BODY]]:
; I32-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; I32-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; I32-NEXT: [[STEP_ADD:%.*]] = add nuw <4 x i32> [[VEC_IND]], splat (i32 4)
-; I32-NEXT: [[STEP_ADD_2:%.*]] = add nuw <4 x i32> [[STEP_ADD]], splat (i32 4)
-; I32-NEXT: [[STEP_ADD_3:%.*]] = add nuw <4 x i32> [[STEP_ADD_2]], splat (i32 4)
+; I32-NEXT: [[VEC_IND:%.*]] = phi <8 x i32> [ <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
; I32-NEXT: [[TMP3:%.*]] = add i32 [[INDEX]], 1
; I32-NEXT: [[TMP4:%.*]] = add i32 [[INDEX]], 2
; I32-NEXT: [[TMP5:%.*]] = add i32 [[INDEX]], 3
@@ -189,131 +183,46 @@ define void @test_store_initially_interleave(i32 %n, ptr noalias %src) #0 {
; I32-NEXT: [[TMP7:%.*]] = add i32 [[INDEX]], 5
; I32-NEXT: [[TMP8:%.*]] = add i32 [[INDEX]], 6
; I32-NEXT: [[TMP9:%.*]] = add i32 [[INDEX]], 7
-; I32-NEXT: [[TMP10:%.*]] = add i32 [[INDEX]], 8
-; I32-NEXT: [[TMP11:%.*]] = add i32 [[INDEX]], 9
-; I32-NEXT: [[TMP12:%.*]] = add i32 [[INDEX]], 10
-; I32-NEXT: [[TMP13:%.*]] = add i32 [[INDEX]], 11
-; I32-NEXT: [[TMP14:%.*]] = add i32 [[INDEX]], 12
-; I32-NEXT: [[TMP15:%.*]] = add i32 [[INDEX]], 13
-; I32-NEXT: [[TMP16:%.*]] = add i32 [[INDEX]], 14
-; I32-NEXT: [[TMP17:%.*]] = add i32 [[INDEX]], 15
-; I32-NEXT: [[TMP18:%.*]] = uitofp <4 x i32> [[VEC_IND]] to <4 x double>
-; I32-NEXT: [[TMP23:%.*]] = uitofp <4 x i32> [[STEP_ADD]] to <4 x double>
-; I32-NEXT: [[TMP28:%.*]] = uitofp <4 x i32> [[STEP_ADD_2]] to <4 x double>
-; I32-NEXT: [[TMP33:%.*]] = uitofp <4 x i32> [[STEP_ADD_3]] to <4 x double>
-; I32-NEXT: [[TMP53:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[INDEX]]
-; I32-NEXT: [[TMP54:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP3]]
-; I32-NEXT: [[TMP55:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP4]]
-; I32-NEXT: [[TMP56:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP5]]
-; I32-NEXT: [[TMP57:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP6]]
-; I32-NEXT: [[TMP58:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP7]]
-; I32-NEXT: [[TMP59:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP8]]
-; I32-NEXT: [[TMP60:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP9]]
-; I32-NEXT: [[TMP61:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP10]]
-; I32-NEXT: [[TMP62:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP11]]
-; I32-NEXT: [[TMP63:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP12]]
-; I32-NEXT: [[TMP64:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP13]]
-; I32-NEXT: [[TMP65:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP14]]
-; I32-NEXT: [[TMP66:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP15]]
-; I32-NEXT: [[TMP67:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP16]]
-; I32-NEXT: [[TMP68:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP17]]
-; I32-NEXT: [[TMP38:%.*]] = load ptr, ptr [[TMP53]], align 4
-; I32-NEXT: [[TMP39:%.*]] = load ptr, ptr [[TMP54]], align 4
-; I32-NEXT: [[TMP40:%.*]] = load ptr, ptr [[TMP55]], align 4
-; I32-NEXT: [[TMP41:%.*]] = load ptr, ptr [[TMP56]], align 4
-; I32-NEXT: [[TMP42:%.*]] = load ptr, ptr [[TMP57]], align 4
-; I32-NEXT: [[TMP43:%.*]] = load ptr, ptr [[TMP58]], align 4
-; I32-NEXT: [[TMP44:%.*]] = load ptr, ptr [[TMP59]], align 4
-; I32-NEXT: [[TMP45:%.*]] = load ptr, ptr [[TMP60]], align 4
-; I32-NEXT: [[TMP46:%.*]] = load ptr, ptr [[TMP61]], align 4
-; I32-NEXT: [[TMP47:%.*]] = load ptr, ptr [[TMP62]], align 4
-; I32-NEXT: [[TMP48:%.*]] = load ptr, ptr [[TMP63]], align 4
-; I32-NEXT: [[TMP49:%.*]] = load ptr, ptr [[TMP64]], align 4
+; I32-NEXT: [[TMP11:%.*]] = uitofp <8 x i32> [[VEC_IND]] to <8 x double>
+; I32-NEXT: [[TMP65:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[INDEX]]
+; I32-NEXT: [[TMP66:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP3]]
+; I32-NEXT: [[TMP67:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP4]]
+; I32-NEXT: [[TMP68:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP5]]
+; I32-NEXT: [[TMP84:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP6]]
+; I32-NEXT: [[TMP85:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP7]]
+; I32-NEXT: [[TMP86:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP8]]
+; I32-NEXT: [[TMP87:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP9]]
; I32-NEXT: [[TMP50:%.*]] = load ptr, ptr [[TMP65]], align 4
; I32-NEXT: [[TMP51:%.*]] = load ptr, ptr [[TMP66]], align 4
; I32-NEXT: [[TMP52:%.*]] = load ptr, ptr [[TMP67]], align 4
; I32-NEXT: [[TMP69:%.*]] = load ptr, ptr [[TMP68]], align 4
-; I32-NEXT: [[TMP19:%.*]] = extractelement <4 x double> [[TMP18]], i64 0
-; I32-NEXT: store double [[TMP19]], ptr [[TMP38]], align 4
-; I32-NEXT: [[TMP20:%.*]] = extractelement <4 x double> [[TMP18]], i64 1
-; I32-NEXT: store double [[TMP20]], ptr [[TMP39]], align 4
-; I32-NEXT: [[TMP21:%.*]] = extractelement <4 x double> [[TMP18]], i64 2
-; I32-NEXT: store double [[TMP21]], ptr [[TMP40]], align 4
-; I32-NEXT: [[TMP22:%.*]] = extractelement <4 x double> [[TMP18]], i64 3
-; I32-NEXT: store double [[TMP22]], ptr [[TMP41]], align 4
-; I32-NEXT: [[TMP24:%.*]] = extractelement <4 x double> [[TMP23]], i64 0
-; I32-NEXT: store double [[TMP24]], ptr [[TMP42]], align 4
-; I32-NEXT: [[TMP25:%.*]] = extractelement <4 x double> [[TMP23]], i64 1
-; I32-NEXT: store double [[TMP25]], ptr [[TMP43]], align 4
-; I32-NEXT: [[TMP26:%.*]] = extractelement <4 x double> [[TMP23]], i64 2
-; I32-NEXT: store double [[TMP26]], ptr [[TMP44]], align 4
-; I32-NEXT: [[TMP27:%.*]] = extractelement <4 x double> [[TMP23]], i64 3
-; I32-NEXT: store double [[TMP27]], ptr [[TMP45]], align 4
-; I32-NEXT: [[TMP29:%.*]] = extractelement <4 x double> [[TMP28]], i64 0
-; I32-NEXT: store double [[TMP29]], ptr [[TMP46]], align 4
-; I32-NEXT: [[TMP30:%.*]] = extractelement <4 x double> [[TMP28]], i64 1
-; I32-NEXT: store double [[TMP30]], ptr [[TMP47]], align 4
-; I32-NEXT: [[TMP31:%.*]] = extractelement <4 x double> [[TMP28]], i64 2
-; I32-NEXT: store double [[TMP31]], ptr [[TMP48]], align 4
-; I32-NEXT: [[TMP32:%.*]] = extractelement <4 x double> [[TMP28]], i64 3
-; I32-NEXT: store double [[TMP32]], ptr [[TMP49]], align 4
-; I32-NEXT: [[TMP34:%.*]] = extractelement <4 x double> [[TMP33]], i64 0
-; I32-NEXT: store double [[TMP34]], ptr [[TMP50]], align 4
-; I32-NEXT: [[TMP35:%.*]] = extractelement <4 x double> [[TMP33]], i64 1
-; I32-NEXT: store double [[TMP35]], ptr [[TMP51]], align 4
-; I32-NEXT: [[TMP36:%.*]] = extractelement <4 x double> [[TMP33]], i64 2
-; I32-NEXT: store double [[TMP36]], ptr [[TMP52]], align 4
-; I32-NEXT: [[TMP37:%.*]] = extractelement <4 x double> [[TMP33]], i64 3
-; I32-NEXT: store double [[TMP37]], ptr [[TMP69]], align 4
-; I32-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 16
-; I32-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[STEP_ADD_3]], splat (i32 4)
-; I32-NEXT: [[TMP70:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
-; I32-NEXT: br i1 [[TMP70]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; I32: [[MIDDLE_BLOCK]]:
-; I32-NEXT: br label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; I32: [[VEC_EPILOG_ITER_CHECK]]:
-; I32-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ule i32 [[TMP2]], 4
-; I32-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3:![0-9]+]]
-; I32: [[VEC_EPILOG_PH]]:
-; I32-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; I32-NEXT: [[N_MOD_VF2:%.*]] = and i32 [[TMP0]], 3
-; I32-NEXT: [[TMP71:%.*]] = icmp eq i32 [[N_MOD_VF2]], 0
-; I32-NEXT: [[TMP72:%.*]] = select i1 [[TMP71]], i32 4, i32 [[N_MOD_VF2]]
-; I32-NEXT: [[N_VEC3:%.*]] = sub i32 [[TMP0]], [[TMP72]]
-; I32-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[VEC_EPILOG_RESUME_VAL]], i64 0
-; I32-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
-; I32-NEXT: [[INDUCTION:%.*]] = add <4 x i32> [[BROADCAST_SPLAT]], <i32 0, i32 1, i32 2, i32 3>
-; I32-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; I32: [[VEC_EPILOG_VECTOR_BODY]]:
-; I32-NEXT: [[INDEX4:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; I32-NEXT: [[VEC_IND5:%.*]] = phi <4 x i32> [ [[INDUCTION]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT7:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; I32-NEXT: [[TMP73:%.*]] = add i32 [[INDEX4]], 1
-; I32-NEXT: [[TMP74:%.*]] = add i32 [[INDEX4]], 2
-; I32-NEXT: [[TMP75:%.*]] = add i32 [[INDEX4]], 3
-; I32-NEXT: [[TMP76:%.*]] = uitofp <4 x i32> [[VEC_IND5]] to <4 x double>
-; I32-NEXT: [[TMP84:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[INDEX4]]
-; I32-NEXT: [[TMP85:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP73]]
-; I32-NEXT: [[TMP86:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP74]]
-; I32-NEXT: [[TMP87:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP75]]
; I32-NEXT: [[TMP81:%.*]] = load ptr, ptr [[TMP84]], align 4
; I32-NEXT: [[TMP82:%.*]] = load ptr, ptr [[TMP85]], align 4
; I32-NEXT: [[TMP83:%.*]] = load ptr, ptr [[TMP86]], align 4
; I32-NEXT: [[TMP88:%.*]] = load ptr, ptr [[TMP87]], align 4
-; I32-NEXT: [[TMP77:%.*]] = extractelement <4 x double> [[TMP76]], i64 0
+; I32-NEXT: [[TMP28:%.*]] = extractelement <8 x double> [[TMP11]], i64 0
+; I32-NEXT: store double [[TMP28]], ptr [[TMP50]], align 4
+; I32-NEXT: [[TMP29:%.*]] = extractelement <8 x double> [[TMP11]], i64 1
+; I32-NEXT: store double [[TMP29]], ptr [[TMP51]], align 4
+; I32-NEXT: [[TMP30:%.*]] = extractelement <8 x double> [[TMP11]], i64 2
+; I32-NEXT: store double [[TMP30]], ptr [[TMP52]], align 4
+; I32-NEXT: [[TMP31:%.*]] = extractelement <8 x double> [[TMP11]], i64 3
+; I32-NEXT: store double [[TMP31]], ptr [[TMP69]], align 4
+; I32-NEXT: [[TMP77:%.*]] = extractelement <8 x double> [[TMP11]], i64 4
; I32-NEXT: store double [[TMP77]], ptr [[TMP81]], align 4
-; I32-NEXT: [[TMP78:%.*]] = extractelement <4 x double> [[TMP76]], i64 1
+; I32-NEXT: [[TMP78:%.*]] = extractelement <8 x double> [[TMP11]], i64 5
; I32-NEXT: store double [[TMP78]], ptr [[TMP82]], align 4
-; I32-NEXT: [[TMP79:%.*]] = extractelement <4 x double> [[TMP76]], i64 2
+; I32-NEXT: [[TMP79:%.*]] = extractelement <8 x double> [[TMP11]], i64 6
; I32-NEXT: store double [[TMP79]], ptr [[TMP83]], align 4
-; I32-NEXT: [[TMP80:%.*]] = extractelement <4 x double> [[TMP76]], i64 3
+; I32-NEXT: [[TMP80:%.*]] = extractelement <8 x double> [[TMP11]], i64 7
; I32-NEXT: store double [[TMP80]], ptr [[TMP88]], align 4
-; I32-NEXT: [[INDEX_NEXT6]] = add nuw i32 [[INDEX4]], 4
-; I32-NEXT: [[VEC_IND_NEXT7]] = add <4 x i32> [[VEC_IND5]], splat (i32 4)
-; I32-NEXT: [[TMP89:%.*]] = icmp eq i32 [[INDEX_NEXT6]], [[N_VEC3]]
-; I32-NEXT: br i1 [[TMP89]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; I32-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 8
+; I32-NEXT: [[VEC_IND_NEXT]] = add <8 x i32> [[VEC_IND]], splat (i32 8)
+; I32-NEXT: [[TMP36:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; I32-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; I32: [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; I32-NEXT: br label %[[VEC_EPILOG_SCALAR_PH]]
-; I32: [[VEC_EPILOG_SCALAR_PH]]:
+; I32-NEXT: br label %[[VEC_EPILOG_PH]]
+; I32: [[VEC_EPILOG_PH]]:
;
entry:
br label %loop
@@ -419,7 +328,7 @@ define void @test_store_loaded_value(ptr noalias %src, ptr noalias %dst, i32 %n)
; I32-NEXT: store double [[TMP10]], ptr [[TMP18]], align 8
; I32-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; I32-NEXT: [[TMP19:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; I32-NEXT: br i1 [[TMP19]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; I32-NEXT: br i1 [[TMP19]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
; I32: [[MIDDLE_BLOCK]]:
; I32-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N_EXT]], [[N_VEC]]
; I32-NEXT: br i1 [[CMP_N]], [[EXIT_LOOPEXIT:label %.*]], label %[[SCALAR_PH]]
@@ -817,7 +726,7 @@ define void @loaded_address_used_by_load_through_blend(i64 %start, ptr noalias %
; I32-NEXT: store float [[TMP82]], ptr [[TMP90]], align 4
; I32-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; I32-NEXT: [[TMP91:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; I32-NEXT: br i1 [[TMP91]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
+; I32-NEXT: br i1 [[TMP91]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
; I32: [[MIDDLE_BLOCK]]:
; I32-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP1]], [[N_VEC]]
; I32-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
@@ -868,54 +777,102 @@ define void @address_use_in_different_block(ptr noalias %dst, ptr %src.0, ptr %s
; I64-NEXT: [[TMP0:%.*]] = add i64 [[INDEX]], 1
; I64-NEXT: [[TMP1:%.*]] = add i64 [[INDEX]], 2
; I64-NEXT: [[TMP2:%.*]] = add i64 [[INDEX]], 3
+; I64-NEXT: [[TMP3:%.*]] = add i64 [[INDEX]], 4
+; I64-NEXT: [[TMP4:%.*]] = add i64 [[INDEX]], 5
+; I64-NEXT: [[TMP5:%.*]] = add i64 [[INDEX]], 6
+; I64-NEXT: [[TMP6:%.*]] = add i64 [[INDEX]], 7
; I64-NEXT: [[TMP11:%.*]] = mul i64 [[INDEX]], [[OFFSET]]
; I64-NEXT: [[TMP12:%.*]] = mul i64 [[TMP0]], [[OFFSET]]
; I64-NEXT: [[TMP13:%.*]] = mul i64 [[TMP1]], [[OFFSET]]
; I64-NEXT: [[TMP14:%.*]] = mul i64 [[TMP2]], [[OFFSET]]
+; I64-NEXT: [[TMP15:%.*]] = mul i64 [[TMP3]], [[OFFSET]]
+; I64-NEXT: [[TMP16:%.*]] = mul i64 [[TMP4]], [[OFFSET]]
+; I64-NEXT: [[TMP17:%.*]] = mul i64 [[TMP5]], [[OFFSET]]
+; I64-NEXT: [[TMP18:%.*]] = mul i64 [[TMP6]], [[OFFSET]]
; I64-NEXT: [[TMP19:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP11]]
; I64-NEXT: [[TMP20:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP12]]
; I64-NEXT: [[TMP21:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP13]]
; I64-NEXT: [[TMP22:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP14]]
+; I64-NEXT: [[TMP23:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP15]]
+; I64-NEXT: [[TMP24:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP16]]
+; I64-NEXT: [[TMP25:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP17]]
+; I64-NEXT: [[TMP26:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP18]]
; I64-NEXT: [[TMP27:%.*]] = load i32, ptr [[TMP19]], align 4
; I64-NEXT: [[TMP28:%.*]] = load i32, ptr [[TMP20]], align 4
; I64-NEXT: [[TMP29:%.*]] = load i32, ptr [[TMP21]], align 4
; I64-NEXT: [[TMP30:%.*]] = load i32, ptr [[TMP22]], align 4
+; I64-NEXT: [[TMP31:%.*]] = load i32, ptr [[TMP23]], align 4
+; I64-NEXT: [[TMP32:%.*]] = load i32, ptr [[TMP24]], align 4
+; I64-NEXT: [[TMP33:%.*]] = load i32, ptr [[TMP25]], align 4
+; I64-NEXT: [[TMP34:%.*]] = load i32, ptr [[TMP26]], align 4
; I64-NEXT: [[TMP35:%.*]] = sext i32 [[TMP27]] to i64
; I64-NEXT: [[TMP36:%.*]] = sext i32 [[TMP28]] to i64
; I64-NEXT: [[TMP37:%.*]] = sext i32 [[TMP29]] to i64
; I64-NEXT: [[TMP38:%.*]] = sext i32 [[TMP30]] to i64
+; I64-NEXT: [[TMP39:%.*]] = sext i32 [[TMP31]] to i64
+; I64-NEXT: [[TMP40:%.*]] = sext i32 [[TMP32]] to i64
+; I64-NEXT: [[TMP41:%.*]] = sext i32 [[TMP33]] to i64
+; I64-NEXT: [[TMP42:%.*]] = sext i32 [[TMP34]] to i64
; I64-NEXT: [[TMP43:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP35]]
; I64-NEXT: [[TMP44:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP36]]
; I64-NEXT: [[TMP45:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP37]]
; I64-NEXT: [[TMP46:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP38]]
+; I64-NEXT: [[TMP47:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP39]]
+; I64-NEXT: [[TMP48:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP40]]
+; I64-NEXT: [[TMP49:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP41]]
+; I64-NEXT: [[TMP50:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP42]]
; I64-NEXT: [[TMP51:%.*]] = getelementptr i8, ptr [[TMP43]], i64 -8
; I64-NEXT: [[TMP52:%.*]] = getelementptr i8, ptr [[TMP44]], i64 -8
; I64-NEXT: [[TMP53:%.*]] = getelementptr i8, ptr [[TMP45]], i64 -8
; I64-NEXT: [[TMP54:%.*]] = getelementptr i8, ptr [[TMP46]], i64 -8
+; I64-NEXT: [[TMP55:%.*]] = getelementptr i8, ptr [[TMP47]], i64 -8
+; I64-NEXT: [[TMP56:%.*]] = getelementptr i8, ptr [[TMP48]], i64 -8
+; I64-NEXT: [[TMP57:%.*]] = getelementptr i8, ptr [[TMP49]], i64 -8
+; I64-NEXT: [[TMP58:%.*]] = getelementptr i8, ptr [[TMP50]], i64 -8
; I64-NEXT: [[TMP63:%.*]] = load double, ptr [[TMP51]], align 8
; I64-NEXT: [[TMP64:%.*]] = load double, ptr [[TMP52]], align 8
; I64-NEXT: [[TMP67:%.*]] = load double, ptr [[TMP53]], align 8
; I64-NEXT: [[TMP68:%.*]] = load double, ptr [[TMP54]], align 8
-; I64-NEXT: [[TMP31:%.*]] = insertelement <4 x double> poison, double [[TMP63]], i64 0
-; I64-NEXT: [[TMP32:%.*]] = insertelement <4 x double> [[TMP31]], double [[TMP64]], i64 1
-; I64-NEXT: [[TMP33:%.*]] = insertelement <4 x double> [[TMP32]], double [[TMP67]], i64 2
-; I64-NEXT: [[TMP34:%.*]] = insertelement <4 x double> [[TMP33]], double [[TMP68]], i64 3
-; I64-NEXT: [[TMP39:%.*]] = fsub <4 x double> zeroinitializer, [[TMP34]]
+; I64-NEXT: [[TMP59:%.*]] = load double, ptr [[TMP55]], align 8
+; I64-NEXT: [[TMP60:%.*]] = load double, ptr [[TMP56]], align 8
+; I64-NEXT: [[TMP61:%.*]] = load double, ptr [[TMP57]], align 8
+; I64-NEXT: [[TMP62:%.*]] = load double, ptr [[TMP58]], align 8
+; I64-NEXT: [[TMP72:%.*]] = insertelement <8 x double> poison, double [[TMP63]], i64 0
+; I64-NEXT: [[TMP73:%.*]] = insertelement <8 x double> [[TMP72]], double [[TMP64]], i64 1
+; I64-NEXT: [[TMP65:%.*]] = insertelement <8 x double> [[TMP73]], double [[TMP67]], i64 2
+; I64-NEXT: [[TMP66:%.*]] = insertelement <8 x double> [[TMP65]], double [[TMP68]], i64 3
+; I64-NEXT: [[TMP74:%.*]] = insertelement <8 x double> [[TMP66]], double [[TMP59]], i64 4
+; I64-NEXT: [[TMP75:%.*]] = insertelement <8 x double> [[TMP74]], double [[TMP60]], i64 5
+; I64-NEXT: [[TMP69:%.*]] = insertelement <8 x double> [[TMP75]], double [[TMP61]], i64 6
+; I64-NEXT: [[TMP70:%.*]] = insertelement <8 x double> [[TMP69]], double [[TMP62]], i64 7
+; I64-NEXT: [[TMP71:%.*]] = fsub <8 x double> zeroinitializer, [[TMP70]]
; I64-NEXT: [[TMP87:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP11]]
; I64-NEXT: [[TMP88:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP12]]
; I64-NEXT: [[TMP89:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP13]]
; I64-NEXT: [[TMP90:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP14]]
-; I64-NEXT: [[TMP78:%.*]] = extractelement <4 x double> [[TMP39]], i64 0
+; I64-NEXT: [[TMP76:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP15]]
+; I64-NEXT: [[TMP77:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP16]]
+; I64-NEXT: [[TMP80:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP17]]
+; I64-NEXT: [[TMP83:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP18]]
+; I64-NEXT: [[TMP78:%.*]] = extractelement <8 x double> [[TMP71]], i64 0
; I64-NEXT: store double [[TMP78]], ptr [[TMP87]], align 8
-; I64-NEXT: [[TMP79:%.*]] = extractelement <4 x double> [[TMP39]], i64 1
+; I64-NEXT: [[TMP79:%.*]] = extractelement <8 x double> [[TMP71]], i64 1
; I64-NEXT: store double [[TMP79]], ptr [[TMP88]], align 8
-; I64-NEXT: [[TMP81:%.*]] = extractelement <4 x double> [[TMP39]], i64 2
+; I64-NEXT: [[TMP81:%.*]] = extractelement <8 x double> [[TMP71]], i64 2
; I64-NEXT: store double [[TMP81]], ptr [[TMP89]], align 8
-; I64-NEXT: [[TMP82:%.*]] = extractelement <4 x double> [[TMP39]], i64 3
+; I64-NEXT: [[TMP82:%.*]] = extractelement <8 x double> [[TMP71]], i64 3
; I64-NEXT: store double [[TMP82]], ptr [[TMP90]], align 8
-; I64-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
-; I64-NEXT: [[TMP47:%.*]] = icmp eq i64 [[INDEX_NEXT]], 100
-; I64-NEXT: br i1 [[TMP47]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; I64-NEXT: [[TMP84:%.*]] = extractelement <8 x double> [[TMP71]], i64 4
+; I64-NEXT: store double [[TMP84]], ptr [[TMP76]], align 8
+; I64-NEXT: [[TMP85:%.*]] = extractelement <8 x double> [[TMP71]], i64 5
+; I64-NEXT: store double [[TMP85]], ptr [[TMP77]], align 8
+; I64-NEXT: [[TMP86:%.*]] = extractelement <8 x double> [[TMP71]], i64 6
+; I64-NEXT: store double [[TMP86]], ptr [[TMP80]], align 8
+; I64-NEXT: [[TMP91:%.*]] = extractelement <8 x double> [[TMP71]], i64 7
+; I64-NEXT: store double [[TMP91]], ptr [[TMP83]], align 8
+; I64-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; I64-NEXT: [[TMP92:%.*]] = icmp eq i64 [[INDEX_NEXT]], 96
+; I64-NEXT: br i1 [[TMP92]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
; I64: [[MIDDLE_BLOCK]]:
; I64-NEXT: br label %[[SCALAR_PH:.*]]
; I64: [[SCALAR_PH]]:
@@ -933,54 +890,102 @@ define void @address_use_in_different_block(ptr noalias %dst, ptr %src.0, ptr %s
; I32-NEXT: [[TMP0:%.*]] = add i64 [[INDEX]], 1
; I32-NEXT: [[TMP1:%.*]] = add i64 [[INDEX]], 2
; I32-NEXT: [[TMP2:%.*]] = add i64 [[INDEX]], 3
+; I32-NEXT: [[TMP31:%.*]] = add i64 [[INDEX]], 4
+; I32-NEXT: [[TMP32:%.*]] = add i64 [[INDEX]], 5
+; I32-NEXT: [[TMP33:%.*]] = add i64 [[INDEX]], 6
+; I32-NEXT: [[TMP34:%.*]] = add i64 [[INDEX]], 7
; I32-NEXT: [[TMP3:%.*]] = mul i64 [[INDEX]], [[OFFSET]]
; I32-NEXT: [[TMP4:%.*]] = mul i64 [[TMP0]], [[OFFSET]]
; I32-NEXT: [[TMP5:%.*]] = mul i64 [[TMP1]], [[OFFSET]]
; I32-NEXT: [[TMP6:%.*]] = mul i64 [[TMP2]], [[OFFSET]]
+; I32-NEXT: [[TMP47:%.*]] = mul i64 [[TMP31]], [[OFFSET]]
+; I32-NEXT: [[TMP48:%.*]] = mul i64 [[TMP32]], [[OFFSET]]
+; I32-NEXT: [[TMP49:%.*]] = mul i64 [[TMP33]], [[OFFSET]]
+; I32-NEXT: [[TMP50:%.*]] = mul i64 [[TMP34]], [[OFFSET]]
; I32-NEXT: [[TMP7:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP3]]
; I32-NEXT: [[TMP8:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP4]]
; I32-NEXT: [[TMP9:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP5]]
; I32-NEXT: [[TMP10:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP6]]
+; I32-NEXT: [[TMP55:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP47]]
+; I32-NEXT: [[TMP56:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP48]]
+; I32-NEXT: [[TMP57:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP49]]
+; I32-NEXT: [[TMP58:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP50]]
; I32-NEXT: [[TMP11:%.*]] = load i32, ptr [[TMP7]], align 4
; I32-NEXT: [[TMP12:%.*]] = load i32, ptr [[TMP8]], align 4
; I32-NEXT: [[TMP13:%.*]] = load i32, ptr [[TMP9]], align 4
; I32-NEXT: [[TMP14:%.*]] = load i32, ptr [[TMP10]], align 4
+; I32-NEXT: [[TMP72:%.*]] = load i32, ptr [[TMP55]], align 4
+; I32-NEXT: [[TMP73:%.*]] = load i32, ptr [[TMP56]], align 4
+; I32-NEXT: [[TMP74:%.*]] = load i32, ptr [[TMP57]], align 4
+; I32-NEXT: [[TMP75:%.*]] = load i32, ptr [[TMP58]], align 4
; I32-NEXT: [[TMP15:%.*]] = sext i32 [[TMP11]] to i64
; I32-NEXT: [[TMP16:%.*]] = sext i32 [[TMP12]] to i64
; I32-NEXT: [[TMP17:%.*]] = sext i32 [[TMP13]] to i64
; I32-NEXT: [[TMP18:%.*]] = sext i32 [[TMP14]] to i64
+; I32-NEXT: [[TMP35:%.*]] = sext i32 [[TMP72]] to i64
+; I32-NEXT: [[TMP80:%.*]] = sext i32 [[TMP73]] to i64
+; I32-NEXT: [[TMP81:%.*]] = sext i32 [[TMP74]] to i64
+; I32-NEXT: [[TMP82:%.*]] = sext i32 [[TMP75]] to i64
; I32-NEXT: [[TMP19:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP15]]
; I32-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP16]]
; I32-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP17]]
; I32-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP18]]
+; I32-NEXT: [[TMP83:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP35]]
+; I32-NEXT: [[TMP44:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP80]]
+; I32-NEXT: [[TMP45:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP81]]
+; I32-NEXT: [[TMP46:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP82]]
; I32-NEXT: [[TMP23:%.*]] = getelementptr i8, ptr [[TMP19]], i64 -8
; I32-NEXT: [[TMP24:%.*]] = getelementptr i8, ptr [[TMP20]], i64 -8
; I32-NEXT: [[TMP25:%.*]] = getelementptr i8, ptr [[TMP21]], i64 -8
; I32-NEXT: [[TMP26:%.*]] = getelementptr i8, ptr [[TMP22]], i64 -8
+; I32-NEXT: [[TMP51:%.*]] = getelementptr i8, ptr [[TMP83]], i64 -8
+; I32-NEXT: [[TMP52:%.*]] = getelementptr i8, ptr [[TMP44]], i64 -8
+; I32-NEXT: [[TMP53:%.*]] = getelementptr i8, ptr [[TMP45]], i64 -8
+; I32-NEXT: [[TMP54:%.*]] = getelementptr i8, ptr [[TMP46]], i64 -8
; I32-NEXT: [[TMP27:%.*]] = load double, ptr [[TMP23]], align 8
; I32-NEXT: [[TMP28:%.*]] = load double, ptr [[TMP24]], align 8
; I32-NEXT: [[TMP29:%.*]] = load double, ptr [[TMP25]], align 8
; I32-NEXT: [[TMP30:%.*]] = load double, ptr [[TMP26]], align 8
-; I32-NEXT: [[TMP31:%.*]] = insertelement <4 x double> poison, double [[TMP27]], i64 0
-; I32-NEXT: [[TMP32:%.*]] = insertelement <4 x double> [[TMP31]], double [[TMP28]], i64 1
-; I32-NEXT: [[TMP33:%.*]] = insertelement <4 x double> [[TMP32]], double [[TMP29]], i64 2
-; I32-NEXT: [[TMP34:%.*]] = insertelement <4 x double> [[TMP33]], double [[TMP30]], i64 3
-; I32-NEXT: [[TMP35:%.*]] = fsub <4 x double> zeroinitializer, [[TMP34]]
+; I32-NEXT: [[TMP59:%.*]] = load double, ptr [[TMP51]], align 8
+; I32-NEXT: [[TMP60:%.*]] = load double, ptr [[TMP52]], align 8
+; I32-NEXT: [[TMP61:%.*]] = load double, ptr [[TMP53]], align 8
+; I32-NEXT: [[TMP62:%.*]] = load double, ptr [[TMP54]], align 8
+; I32-NEXT: [[TMP63:%.*]] = insertelement <8 x double> poison, double [[TMP27]], i64 0
+; I32-NEXT: [[TMP64:%.*]] = insertelement <8 x double> [[TMP63]], double [[TMP28]], i64 1
+; I32-NEXT: [[TMP65:%.*]] = insertelement <8 x double> [[TMP64]], double [[TMP29]], i64 2
+; I32-NEXT: [[TMP66:%.*]] = insertelement <8 x double> [[TMP65]], double [[TMP30]], i64 3
+; I32-NEXT: [[TMP67:%.*]] = insertelement <8 x double> [[TMP66]], double [[TMP59]], i64 4
+; I32-NEXT: [[TMP68:%.*]] = insertelement <8 x double> [[TMP67]], double [[TMP60]], i64 5
+; I32-NEXT: [[TMP69:%.*]] = insertelement <8 x double> [[TMP68]], double [[TMP61]], i64 6
+; I32-NEXT: [[TMP70:%.*]] = insertelement <8 x double> [[TMP69]], double [[TMP62]], i64 7
+; I32-NEXT: [[TMP71:%.*]] = fsub <8 x double> zeroinitializer, [[TMP70]]
; I32-NEXT: [[TMP40:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP3]]
; I32-NEXT: [[TMP41:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP4]]
; I32-NEXT: [[TMP42:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP5]]
; I32-NEXT: [[TMP43:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP6]]
-; I32-NEXT: [[TMP36:%.*]] = extractelement <4 x double> [[TMP35]], i64 0
+; I32-NEXT: [[TMP76:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP47]]
+; I32-NEXT: [[TMP77:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP48]]
+; I32-NEXT: [[TMP78:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP49]]
+; I32-NEXT: [[TMP79:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP50]]
+; I32-NEXT: [[TMP36:%.*]] = extractelement <8 x double> [[TMP71]], i64 0
; I32-NEXT: store double [[TMP36]], ptr [[TMP40]], align 8
-; I32-NEXT: [[TMP37:%.*]] = extractelement <4 x double> [[TMP35]], i64 1
+; I32-NEXT: [[TMP37:%.*]] = extractelement <8 x double> [[TMP71]], i64 1
; I32-NEXT: store double [[TMP37]], ptr [[TMP41]], align 8
-; I32-NEXT: [[TMP38:%.*]] = extractelement <4 x double> [[TMP35]], i64 2
+; I32-NEXT: [[TMP38:%.*]] = extractelement <8 x double> [[TMP71]], i64 2
; I32-NEXT: store double [[TMP38]], ptr [[TMP42]], align 8
-; I32-NEXT: [[TMP39:%.*]] = extractelement <4 x double> [[TMP35]], i64 3
+; I32-NEXT: [[TMP39:%.*]] = extractelement <8 x double> [[TMP71]], i64 3
; I32-NEXT: store double [[TMP39]], ptr [[TMP43]], align 8
-; I32-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
-; I32-NEXT: [[TMP44:%.*]] = icmp eq i64 [[INDEX_NEXT]], 100
-; I32-NEXT: br i1 [[TMP44]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; I32-NEXT: [[TMP84:%.*]] = extractelement <8 x double> [[TMP71]], i64 4
+; I32-NEXT: store double [[TMP84]], ptr [[TMP76]], align 8
+; I32-NEXT: [[TMP85:%.*]] = extractelement <8 x double> [[TMP71]], i64 5
+; I32-NEXT: store double [[TMP85]], ptr [[TMP77]], align 8
+; I32-NEXT: [[TMP86:%.*]] = extractelement <8 x double> [[TMP71]], i64 6
+; I32-NEXT: store double [[TMP86]], ptr [[TMP78]], align 8
+; I32-NEXT: [[TMP87:%.*]] = extractelement <8 x double> [[TMP71]], i64 7
+; I32-NEXT: store double [[TMP87]], ptr [[TMP79]], align 8
+; I32-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; I32-NEXT: [[TMP88:%.*]] = icmp eq i64 [[INDEX_NEXT]], 96
+; I32-NEXT: br i1 [[TMP88]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
; I32: [[MIDDLE_BLOCK]]:
; I32-NEXT: br label %[[SCALAR_PH:.*]]
; I32: [[SCALAR_PH]]:
@@ -1355,7 +1360,7 @@ define void @invariant_pred_store_sunk_out_of_loop(ptr noalias %dst, ptr noalias
; I32-NEXT: [[TMP5]] = add <2 x i64> [[TMP3]], splat (i64 1)
; I32-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; I32-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1000
-; I32-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; I32-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
; I32: [[MIDDLE_BLOCK]]:
; I32-NEXT: [[BIN_RDX:%.*]] = add <2 x i64> [[TMP5]], [[TMP4]]
; I32-NEXT: [[TMP7:%.*]] = call i64 @llvm.vector.reduce.add.v2i64(<2 x i64> [[BIN_RDX]])
diff --git a/llvm/test/Transforms/LoopVectorize/X86/strided_load_cost.ll b/llvm/test/Transforms/LoopVectorize/X86/strided_load_cost.ll
index 576a27bce5df4..4c5bf44ee7911 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/strided_load_cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/strided_load_cost.ll
@@ -509,12 +509,36 @@ define void @test(ptr %A, ptr noalias %B) #0 {
; CHECK-NEXT: [[TMP5:%.*]] = add i64 [[OFFSET_IDX]], 10
; CHECK-NEXT: [[TMP6:%.*]] = add i64 [[OFFSET_IDX]], 12
; CHECK-NEXT: [[TMP7:%.*]] = add i64 [[OFFSET_IDX]], 14
+; CHECK-NEXT: [[TMP8:%.*]] = add i64 [[OFFSET_IDX]], 16
+; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[OFFSET_IDX]], 18
+; CHECK-NEXT: [[TMP10:%.*]] = add i64 [[OFFSET_IDX]], 20
+; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[OFFSET_IDX]], 22
+; CHECK-NEXT: [[TMP12:%.*]] = add i64 [[OFFSET_IDX]], 24
+; CHECK-NEXT: [[TMP13:%.*]] = add i64 [[OFFSET_IDX]], 26
+; CHECK-NEXT: [[TMP14:%.*]] = add i64 [[OFFSET_IDX]], 28
+; CHECK-NEXT: [[TMP15:%.*]] = add i64 [[OFFSET_IDX]], 30
+; CHECK-NEXT: [[TMP37:%.*]] = add i64 [[OFFSET_IDX]], 32
+; CHECK-NEXT: [[TMP17:%.*]] = add i64 [[OFFSET_IDX]], 34
+; CHECK-NEXT: [[TMP18:%.*]] = add i64 [[OFFSET_IDX]], 36
+; CHECK-NEXT: [[TMP19:%.*]] = add i64 [[OFFSET_IDX]], 38
+; CHECK-NEXT: [[TMP38:%.*]] = add i64 [[OFFSET_IDX]], 40
+; CHECK-NEXT: [[TMP39:%.*]] = add i64 [[OFFSET_IDX]], 42
+; CHECK-NEXT: [[TMP40:%.*]] = add i64 [[OFFSET_IDX]], 44
+; CHECK-NEXT: [[TMP41:%.*]] = add i64 [[OFFSET_IDX]], 46
+; CHECK-NEXT: [[TMP42:%.*]] = add i64 [[OFFSET_IDX]], 48
+; CHECK-NEXT: [[TMP67:%.*]] = add i64 [[OFFSET_IDX]], 50
+; CHECK-NEXT: [[TMP68:%.*]] = add i64 [[OFFSET_IDX]], 52
+; CHECK-NEXT: [[TMP69:%.*]] = add i64 [[OFFSET_IDX]], 54
+; CHECK-NEXT: [[TMP70:%.*]] = add i64 [[OFFSET_IDX]], 56
+; CHECK-NEXT: [[TMP71:%.*]] = add i64 [[OFFSET_IDX]], 58
+; CHECK-NEXT: [[TMP72:%.*]] = add i64 [[OFFSET_IDX]], 60
+; CHECK-NEXT: [[TMP73:%.*]] = add i64 [[OFFSET_IDX]], 62
; CHECK-NEXT: [[TMP16:%.*]] = getelementptr inbounds [1024 x i32], ptr [[A]], i64 0, i64 [[OFFSET_IDX]]
-; CHECK-NEXT: [[WIDE_VEC:%.*]] = load <16 x i32>, ptr [[TMP16]], align 4
-; CHECK-NEXT: [[STRIDED_VEC:%.*]] = shufflevector <16 x i32> [[WIDE_VEC]], <16 x i32> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
-; CHECK-NEXT: [[STRIDED_VEC1:%.*]] = shufflevector <16 x i32> [[WIDE_VEC]], <16 x i32> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
-; CHECK-NEXT: [[TMP18:%.*]] = add <8 x i32> [[STRIDED_VEC]], [[STRIDED_VEC1]]
-; CHECK-NEXT: [[TMP19:%.*]] = trunc <8 x i32> [[TMP18]] to <8 x i8>
+; CHECK-NEXT: [[WIDE_VEC:%.*]] = load <64 x i32>, ptr [[TMP16]], align 4
+; CHECK-NEXT: [[STRIDED_VEC:%.*]] = shufflevector <64 x i32> [[WIDE_VEC]], <64 x i32> poison, <32 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30, i32 32, i32 34, i32 36, i32 38, i32 40, i32 42, i32 44, i32 46, i32 48, i32 50, i32 52, i32 54, i32 56, i32 58, i32 60, i32 62>
+; CHECK-NEXT: [[STRIDED_VEC1:%.*]] = shufflevector <64 x i32> [[WIDE_VEC]], <64 x i32> poison, <32 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31, i32 33, i32 35, i32 37, i32 39, i32 41, i32 43, i32 45, i32 47, i32 49, i32 51, i32 53, i32 55, i32 57, i32 59, i32 61, i32 63>
+; CHECK-NEXT: [[TMP74:%.*]] = add <32 x i32> [[STRIDED_VEC]], [[STRIDED_VEC1]]
+; CHECK-NEXT: [[TMP99:%.*]] = trunc <32 x i32> [[TMP74]] to <32 x i8>
; CHECK-NEXT: [[TMP20:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[OFFSET_IDX]]
; CHECK-NEXT: [[TMP21:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP1]]
; CHECK-NEXT: [[TMP22:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP2]]
@@ -523,23 +547,95 @@ define void @test(ptr %A, ptr noalias %B) #0 {
; CHECK-NEXT: [[TMP25:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP5]]
; CHECK-NEXT: [[TMP26:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP6]]
; CHECK-NEXT: [[TMP27:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP7]]
-; CHECK-NEXT: [[TMP28:%.*]] = extractelement <8 x i8> [[TMP19]], i64 0
+; CHECK-NEXT: [[TMP43:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP8]]
+; CHECK-NEXT: [[TMP44:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP9]]
+; CHECK-NEXT: [[TMP45:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP10]]
+; CHECK-NEXT: [[TMP46:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP11]]
+; CHECK-NEXT: [[TMP47:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP12]]
+; CHECK-NEXT: [[TMP48:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP13]]
+; CHECK-NEXT: [[TMP49:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP14]]
+; CHECK-NEXT: [[TMP50:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP15]]
+; CHECK-NEXT: [[TMP51:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP37]]
+; CHECK-NEXT: [[TMP52:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP17]]
+; CHECK-NEXT: [[TMP53:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP18]]
+; CHECK-NEXT: [[TMP54:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP19]]
+; CHECK-NEXT: [[TMP55:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP38]]
+; CHECK-NEXT: [[TMP56:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP39]]
+; CHECK-NEXT: [[TMP57:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP40]]
+; CHECK-NEXT: [[TMP58:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP41]]
+; CHECK-NEXT: [[TMP59:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP42]]
+; CHECK-NEXT: [[TMP60:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP67]]
+; CHECK-NEXT: [[TMP61:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP68]]
+; CHECK-NEXT: [[TMP62:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP69]]
+; CHECK-NEXT: [[TMP63:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP70]]
+; CHECK-NEXT: [[TMP64:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP71]]
+; CHECK-NEXT: [[TMP65:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP72]]
+; CHECK-NEXT: [[TMP66:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP73]]
+; CHECK-NEXT: [[TMP28:%.*]] = extractelement <32 x i8> [[TMP99]], i64 0
; CHECK-NEXT: store i8 [[TMP28]], ptr [[TMP20]], align 1
-; CHECK-NEXT: [[TMP29:%.*]] = extractelement <8 x i8> [[TMP19]], i64 1
+; CHECK-NEXT: [[TMP29:%.*]] = extractelement <32 x i8> [[TMP99]], i64 1
; CHECK-NEXT: store i8 [[TMP29]], ptr [[TMP21]], align 1
-; CHECK-NEXT: [[TMP30:%.*]] = extractelement <8 x i8> [[TMP19]], i64 2
+; CHECK-NEXT: [[TMP30:%.*]] = extractelement <32 x i8> [[TMP99]], i64 2
; CHECK-NEXT: store i8 [[TMP30]], ptr [[TMP22]], align 1
-; CHECK-NEXT: [[TMP31:%.*]] = extractelement <8 x i8> [[TMP19]], i64 3
+; CHECK-NEXT: [[TMP31:%.*]] = extractelement <32 x i8> [[TMP99]], i64 3
; CHECK-NEXT: store i8 [[TMP31]], ptr [[TMP23]], align 1
-; CHECK-NEXT: [[TMP32:%.*]] = extractelement <8 x i8> [[TMP19]], i64 4
+; CHECK-NEXT: [[TMP32:%.*]] = extractelement <32 x i8> [[TMP99]], i64 4
; CHECK-NEXT: store i8 [[TMP32]], ptr [[TMP24]], align 1
-; CHECK-NEXT: [[TMP33:%.*]] = extractelement <8 x i8> [[TMP19]], i64 5
+; CHECK-NEXT: [[TMP33:%.*]] = extractelement <32 x i8> [[TMP99]], i64 5
; CHECK-NEXT: store i8 [[TMP33]], ptr [[TMP25]], align 1
-; CHECK-NEXT: [[TMP34:%.*]] = extractelement <8 x i8> [[TMP19]], i64 6
+; CHECK-NEXT: [[TMP34:%.*]] = extractelement <32 x i8> [[TMP99]], i64 6
; CHECK-NEXT: store i8 [[TMP34]], ptr [[TMP26]], align 1
-; CHECK-NEXT: [[TMP35:%.*]] = extractelement <8 x i8> [[TMP19]], i64 7
+; CHECK-NEXT: [[TMP35:%.*]] = extractelement <32 x i8> [[TMP99]], i64 7
; CHECK-NEXT: store i8 [[TMP35]], ptr [[TMP27]], align 1
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-NEXT: [[TMP75:%.*]] = extractelement <32 x i8> [[TMP99]], i64 8
+; CHECK-NEXT: store i8 [[TMP75]], ptr [[TMP43]], align 1
+; CHECK-NEXT: [[TMP76:%.*]] = extractelement <32 x i8> [[TMP99]], i64 9
+; CHECK-NEXT: store i8 [[TMP76]], ptr [[TMP44]], align 1
+; CHECK-NEXT: [[TMP77:%.*]] = extractelement <32 x i8> [[TMP99]], i64 10
+; CHECK-NEXT: store i8 [[TMP77]], ptr [[TMP45]], align 1
+; CHECK-NEXT: [[TMP78:%.*]] = extractelement <32 x i8> [[TMP99]], i64 11
+; CHECK-NEXT: store i8 [[TMP78]], ptr [[TMP46]], align 1
+; CHECK-NEXT: [[TMP79:%.*]] = extractelement <32 x i8> [[TMP99]], i64 12
+; CHECK-NEXT: store i8 [[TMP79]], ptr [[TMP47]], align 1
+; CHECK-NEXT: [[TMP80:%.*]] = extractelement <32 x i8> [[TMP99]], i64 13
+; CHECK-NEXT: store i8 [[TMP80]], ptr [[TMP48]], align 1
+; CHECK-NEXT: [[TMP81:%.*]] = extractelement <32 x i8> [[TMP99]], i64 14
+; CHECK-NEXT: store i8 [[TMP81]], ptr [[TMP49]], align 1
+; CHECK-NEXT: [[TMP82:%.*]] = extractelement <32 x i8> [[TMP99]], i64 15
+; CHECK-NEXT: store i8 [[TMP82]], ptr [[TMP50]], align 1
+; CHECK-NEXT: [[TMP83:%.*]] = extractelement <32 x i8> [[TMP99]], i64 16
+; CHECK-NEXT: store i8 [[TMP83]], ptr [[TMP51]], align 1
+; CHECK-NEXT: [[TMP84:%.*]] = extractelement <32 x i8> [[TMP99]], i64 17
+; CHECK-NEXT: store i8 [[TMP84]], ptr [[TMP52]], align 1
+; CHECK-NEXT: [[TMP85:%.*]] = extractelement <32 x i8> [[TMP99]], i64 18
+; CHECK-NEXT: store i8 [[TMP85]], ptr [[TMP53]], align 1
+; CHECK-NEXT: [[TMP86:%.*]] = extractelement <32 x i8> [[TMP99]], i64 19
+; CHECK-NEXT: store i8 [[TMP86]], ptr [[TMP54]], align 1
+; CHECK-NEXT: [[TMP87:%.*]] = extractelement <32 x i8> [[TMP99]], i64 20
+; CHECK-NEXT: store i8 [[TMP87]], ptr [[TMP55]], align 1
+; CHECK-NEXT: [[TMP88:%.*]] = extractelement <32 x i8> [[TMP99]], i64 21
+; CHECK-NEXT: store i8 [[TMP88]], ptr [[TMP56]], align 1
+; CHECK-NEXT: [[TMP89:%.*]] = extractelement <32 x i8> [[TMP99]], i64 22
+; CHECK-NEXT: store i8 [[TMP89]], ptr [[TMP57]], align 1
+; CHECK-NEXT: [[TMP90:%.*]] = extractelement <32 x i8> [[TMP99]], i64 23
+; CHECK-NEXT: store i8 [[TMP90]], ptr [[TMP58]], align 1
+; CHECK-NEXT: [[TMP91:%.*]] = extractelement <32 x i8> [[TMP99]], i64 24
+; CHECK-NEXT: store i8 [[TMP91]], ptr [[TMP59]], align 1
+; CHECK-NEXT: [[TMP92:%.*]] = extractelement <32 x i8> [[TMP99]], i64 25
+; CHECK-NEXT: store i8 [[TMP92]], ptr [[TMP60]], align 1
+; CHECK-NEXT: [[TMP93:%.*]] = extractelement <32 x i8> [[TMP99]], i64 26
+; CHECK-NEXT: store i8 [[TMP93]], ptr [[TMP61]], align 1
+; CHECK-NEXT: [[TMP94:%.*]] = extractelement <32 x i8> [[TMP99]], i64 27
+; CHECK-NEXT: store i8 [[TMP94]], ptr [[TMP62]], align 1
+; CHECK-NEXT: [[TMP95:%.*]] = extractelement <32 x i8> [[TMP99]], i64 28
+; CHECK-NEXT: store i8 [[TMP95]], ptr [[TMP63]], align 1
+; CHECK-NEXT: [[TMP96:%.*]] = extractelement <32 x i8> [[TMP99]], i64 29
+; CHECK-NEXT: store i8 [[TMP96]], ptr [[TMP64]], align 1
+; CHECK-NEXT: [[TMP97:%.*]] = extractelement <32 x i8> [[TMP99]], i64 30
+; CHECK-NEXT: store i8 [[TMP97]], ptr [[TMP65]], align 1
+; CHECK-NEXT: [[TMP98:%.*]] = extractelement <32 x i8> [[TMP99]], i64 31
+; CHECK-NEXT: store i8 [[TMP98]], ptr [[TMP66]], align 1
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; CHECK-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT]], 512
; CHECK-NEXT: br i1 [[TMP36]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/vector_ptr_load_store.ll b/llvm/test/Transforms/LoopVectorize/X86/vector_ptr_load_store.ll
index 91907c3e4d69e..fe886aac6f9b5 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/vector_ptr_load_store.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/vector_ptr_load_store.ll
@@ -96,7 +96,7 @@ define void @test_nonconsecutive_store() {
;; pointer types into account.
; CHECK: test_consecutive_ptr_load
; CHECK: LV: The Smallest and Widest types: 8 / 64 bits.
-; CHECK: LV: Selecting VF: 4
+; CHECK: LV: Selecting VF: 16
define i8 @test_consecutive_ptr_load() readonly {
br label %1
@@ -121,7 +121,7 @@ define i8 @test_consecutive_ptr_load() readonly {
;; However, we should not take unconsecutive loads of pointers into account.
; CHECK: test_nonconsecutive_ptr_load
; CHECK: LV: The Smallest and Widest types: 16 / 64 bits.
-; CHECK: LV: Selecting VF: 1
+; CHECK: LV: Selecting VF: 16
define void @test_nonconsecutive_ptr_load() {
br label %1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks-loopid-dbg.ll b/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks-loopid-dbg.ll
index 9ded7fa7d6a3c..ef6a4b37878f8 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks-loopid-dbg.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks-loopid-dbg.ll
@@ -1,14 +1,133 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
; RUN: opt < %s -passes=loop-vectorize -mtriple=x86_64-unknown-linux -S -pass-remarks='loop-vectorize' 2>&1 | FileCheck -check-prefix=VECTORIZED %s
; RUN: opt < %s -passes=loop-vectorize -force-vector-width=1 -force-vector-interleave=4 -mtriple=x86_64-unknown-linux -S -pass-remarks='loop-vectorize' 2>&1 | FileCheck -check-prefix=UNROLLED %s
; RUN: opt < %s -passes=loop-vectorize -force-vector-width=1 -force-vector-interleave=1 -mtriple=x86_64-unknown-linux -S -pass-remarks-analysis='loop-vectorize' 2>&1 | FileCheck -check-prefix=NONE %s
-; VECTORIZED: remark: vectorization-remarks.c:17:8: vectorized loop (vectorization width: 4, interleaved count: 2)
+; VECTORIZED: remark: vectorization-remarks.c:17:8: vectorized loop (vectorization width: 16, interleaved count: 1)
; UNROLLED: remark: vectorization-remarks.c:17:8: interleaved loop (interleaved count: 4)
; NONE: remark: vectorization-remarks.c:17:8: loop not vectorized: vectorization and interleaving are explicitly disabled, or the loop has already been vectorized
target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
define i32 @foo(i32 %n) #0 !dbg !4 {
+; VECTORIZED-LABEL: define i32 @foo(
+; VECTORIZED-SAME: i32 [[N:%.*]]) !dbg [[DBG4:![0-9]+]] {
+; VECTORIZED-NEXT: [[ENTRY:.*:]]
+; VECTORIZED-NEXT: [[DIFF:%.*]] = alloca i32, align 4
+; VECTORIZED-NEXT: [[CB:%.*]] = alloca [16 x i8], align 16
+; VECTORIZED-NEXT: [[CC:%.*]] = alloca [16 x i8], align 16
+; VECTORIZED-NEXT: store i32 0, ptr [[DIFF]], align 4
+; VECTORIZED-NEXT: br label %[[VECTOR_PH:.*]]
+; VECTORIZED: [[VECTOR_PH]]:
+; VECTORIZED-NEXT: br label %[[VECTOR_BODY:.*]]
+; VECTORIZED: [[VECTOR_BODY]]:
+; VECTORIZED-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[CB]], align 1
+; VECTORIZED-NEXT: [[TMP0:%.*]] = sext <16 x i8> [[WIDE_LOAD]] to <16 x i32>
+; VECTORIZED-NEXT: [[WIDE_LOAD1:%.*]] = load <16 x i8>, ptr [[CC]], align 1
+; VECTORIZED-NEXT: [[TMP1:%.*]] = sext <16 x i8> [[WIDE_LOAD1]] to <16 x i32>
+; VECTORIZED-NEXT: [[TMP2:%.*]] = sub <16 x i32> [[TMP0]], [[TMP1]]
+; VECTORIZED-NEXT: [[TMP3:%.*]] = add <16 x i32> [[TMP2]], zeroinitializer
+; VECTORIZED-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; VECTORIZED: [[MIDDLE_BLOCK]]:
+; VECTORIZED-NEXT: [[TMP4:%.*]] = call i32 @llvm.vector.reduce.add.v16i32(<16 x i32> [[TMP3]])
+; VECTORIZED-NEXT: br label %[[FOR_END:.*]]
+; VECTORIZED: [[FOR_END]]:
+; VECTORIZED-NEXT: store i32 [[TMP4]], ptr [[DIFF]], align 4
+; VECTORIZED-NEXT: call void @ibar(ptr [[DIFF]])
+; VECTORIZED-NEXT: ret i32 0
+;
+; UNROLLED-LABEL: define i32 @foo(
+; UNROLLED-SAME: i32 [[N:%.*]]) !dbg [[DBG4:![0-9]+]] {
+; UNROLLED-NEXT: [[ENTRY:.*:]]
+; UNROLLED-NEXT: [[DIFF:%.*]] = alloca i32, align 4
+; UNROLLED-NEXT: [[CB:%.*]] = alloca [16 x i8], align 16
+; UNROLLED-NEXT: [[CC:%.*]] = alloca [16 x i8], align 16
+; UNROLLED-NEXT: store i32 0, ptr [[DIFF]], align 4
+; UNROLLED-NEXT: br label %[[VECTOR_PH:.*]]
+; UNROLLED: [[VECTOR_PH]]:
+; UNROLLED-NEXT: br label %[[VECTOR_BODY:.*]]
+; UNROLLED: [[VECTOR_BODY]]:
+; UNROLLED-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[VEC_PHI:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP31:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[VEC_PHI1:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP32:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[VEC_PHI2:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP33:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[VEC_PHI3:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP34:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[TMP0:%.*]] = add i64 [[INDEX]], 1
+; UNROLLED-NEXT: [[TMP1:%.*]] = add i64 [[INDEX]], 2
+; UNROLLED-NEXT: [[TMP2:%.*]] = add i64 [[INDEX]], 3
+; UNROLLED-NEXT: [[TMP3:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[INDEX]]
+; UNROLLED-NEXT: [[TMP4:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[TMP0]]
+; UNROLLED-NEXT: [[TMP5:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[TMP1]]
+; UNROLLED-NEXT: [[TMP6:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[TMP2]]
+; UNROLLED-NEXT: [[TMP7:%.*]] = load i8, ptr [[TMP3]], align 1
+; UNROLLED-NEXT: [[TMP8:%.*]] = load i8, ptr [[TMP4]], align 1
+; UNROLLED-NEXT: [[TMP9:%.*]] = load i8, ptr [[TMP5]], align 1
+; UNROLLED-NEXT: [[TMP10:%.*]] = load i8, ptr [[TMP6]], align 1
+; UNROLLED-NEXT: [[TMP11:%.*]] = sext i8 [[TMP7]] to i32
+; UNROLLED-NEXT: [[TMP12:%.*]] = sext i8 [[TMP8]] to i32
+; UNROLLED-NEXT: [[TMP13:%.*]] = sext i8 [[TMP9]] to i32
+; UNROLLED-NEXT: [[TMP14:%.*]] = sext i8 [[TMP10]] to i32
+; UNROLLED-NEXT: [[TMP15:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[INDEX]]
+; UNROLLED-NEXT: [[TMP16:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[TMP0]]
+; UNROLLED-NEXT: [[TMP17:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[TMP1]]
+; UNROLLED-NEXT: [[TMP18:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[TMP2]]
+; UNROLLED-NEXT: [[TMP19:%.*]] = load i8, ptr [[TMP15]], align 1
+; UNROLLED-NEXT: [[TMP20:%.*]] = load i8, ptr [[TMP16]], align 1
+; UNROLLED-NEXT: [[TMP21:%.*]] = load i8, ptr [[TMP17]], align 1
+; UNROLLED-NEXT: [[TMP22:%.*]] = load i8, ptr [[TMP18]], align 1
+; UNROLLED-NEXT: [[TMP23:%.*]] = sext i8 [[TMP19]] to i32
+; UNROLLED-NEXT: [[TMP24:%.*]] = sext i8 [[TMP20]] to i32
+; UNROLLED-NEXT: [[TMP25:%.*]] = sext i8 [[TMP21]] to i32
+; UNROLLED-NEXT: [[TMP26:%.*]] = sext i8 [[TMP22]] to i32
+; UNROLLED-NEXT: [[TMP27:%.*]] = sub i32 [[TMP11]], [[TMP23]]
+; UNROLLED-NEXT: [[TMP28:%.*]] = sub i32 [[TMP12]], [[TMP24]]
+; UNROLLED-NEXT: [[TMP29:%.*]] = sub i32 [[TMP13]], [[TMP25]]
+; UNROLLED-NEXT: [[TMP30:%.*]] = sub i32 [[TMP14]], [[TMP26]]
+; UNROLLED-NEXT: [[TMP31]] = add i32 [[TMP27]], [[VEC_PHI]]
+; UNROLLED-NEXT: [[TMP32]] = add i32 [[TMP28]], [[VEC_PHI1]]
+; UNROLLED-NEXT: [[TMP33]] = add i32 [[TMP29]], [[VEC_PHI2]]
+; UNROLLED-NEXT: [[TMP34]] = add i32 [[TMP30]], [[VEC_PHI3]]
+; UNROLLED-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; UNROLLED-NEXT: [[TMP35:%.*]] = icmp eq i64 [[INDEX_NEXT]], 16
+; UNROLLED-NEXT: br i1 [[TMP35]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; UNROLLED: [[MIDDLE_BLOCK]]:
+; UNROLLED-NEXT: [[BIN_RDX:%.*]] = add i32 [[TMP32]], [[TMP31]]
+; UNROLLED-NEXT: [[BIN_RDX4:%.*]] = add i32 [[TMP33]], [[BIN_RDX]]
+; UNROLLED-NEXT: [[BIN_RDX5:%.*]] = add i32 [[TMP34]], [[BIN_RDX4]]
+; UNROLLED-NEXT: br label %[[FOR_END:.*]]
+; UNROLLED: [[FOR_END]]:
+; UNROLLED-NEXT: store i32 [[BIN_RDX5]], ptr [[DIFF]], align 4
+; UNROLLED-NEXT: call void @ibar(ptr [[DIFF]])
+; UNROLLED-NEXT: ret i32 0
+;
+; NONE-LABEL: define i32 @foo(
+; NONE-SAME: i32 [[N:%.*]]) !dbg [[DBG4:![0-9]+]] {
+; NONE-NEXT: [[ENTRY:.*]]:
+; NONE-NEXT: [[DIFF:%.*]] = alloca i32, align 4
+; NONE-NEXT: [[CB:%.*]] = alloca [16 x i8], align 16
+; NONE-NEXT: [[CC:%.*]] = alloca [16 x i8], align 16
+; NONE-NEXT: store i32 0, ptr [[DIFF]], align 4
+; NONE-NEXT: br label %[[FOR_BODY:.*]]
+; NONE: [[FOR_BODY]]:
+; NONE-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDVARS_IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; NONE-NEXT: [[ADD8:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ADD:%.*]], %[[FOR_BODY]] ]
+; NONE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[INDVARS_IV]]
+; NONE-NEXT: [[TMP0:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
+; NONE-NEXT: [[CONV:%.*]] = sext i8 [[TMP0]] to i32
+; NONE-NEXT: [[ARRAYIDX2:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[INDVARS_IV]]
+; NONE-NEXT: [[TMP1:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
+; NONE-NEXT: [[CONV3:%.*]] = sext i8 [[TMP1]] to i32
+; NONE-NEXT: [[SUB:%.*]] = sub i32 [[CONV]], [[CONV3]]
+; NONE-NEXT: [[ADD]] = add nsw i32 [[SUB]], [[ADD8]]
+; NONE-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1
+; NONE-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], 16
+; NONE-NEXT: br i1 [[EXITCOND]], label %[[FOR_END:.*]], label %[[FOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; NONE: [[FOR_END]]:
+; NONE-NEXT: [[ADD_LCSSA:%.*]] = phi i32 [ [[ADD]], %[[FOR_BODY]] ]
+; NONE-NEXT: store i32 [[ADD_LCSSA]], ptr [[DIFF]], align 4
+; NONE-NEXT: call void @ibar(ptr [[DIFF]])
+; NONE-NEXT: ret i32 0
+;
entry:
%diff = alloca i32, align 4
%cb = alloca [16 x i8], align 16
@@ -61,3 +180,31 @@ declare void @ibar(ptr) #1
!23 = !DILocation(line: 21, column: 3, scope: !4)
!24 = distinct !DICompileUnit(language: DW_LANG_C89, file: !1, emissionKind: NoDebug)
!25 = !{!25, !15}
+;.
+; VECTORIZED: [[META2:![0-9]+]] = distinct !DICompileUnit(language: DW_LANG_C89, file: [[META3:![0-9]+]], isOptimized: false, runtimeVersion: 0, emissionKind: NoDebug)
+; VECTORIZED: [[META3]] = !DIFile(filename: "{{.*}}vectorization-remarks.c", directory: {{.*}})
+; VECTORIZED: [[DBG4]] = distinct !DISubprogram(name: "foo", scope: [[META3]], file: [[META3]], line: 5, type: [[META5:![0-9]+]], scopeLine: 6, virtualIndex: 6, flags: DIFlagPrototyped, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META2]], retainedNodes: [[META6:![0-9]+]])
+; VECTORIZED: [[META5]] = !DISubroutineType(types: [[META6]])
+; VECTORIZED: [[META6]] = !{}
+;.
+; UNROLLED: [[META2:![0-9]+]] = distinct !DICompileUnit(language: DW_LANG_C89, file: [[META3:![0-9]+]], isOptimized: false, runtimeVersion: 0, emissionKind: NoDebug)
+; UNROLLED: [[META3]] = !DIFile(filename: "{{.*}}vectorization-remarks.c", directory: {{.*}})
+; UNROLLED: [[DBG4]] = distinct !DISubprogram(name: "foo", scope: [[META3]], file: [[META3]], line: 5, type: [[META5:![0-9]+]], scopeLine: 6, virtualIndex: 6, flags: DIFlagPrototyped, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META2]], retainedNodes: [[META6:![0-9]+]])
+; UNROLLED: [[META5]] = !DISubroutineType(types: [[META6]])
+; UNROLLED: [[META6]] = !{}
+; UNROLLED: [[LOOP7]] = distinct !{[[LOOP7]], [[META8:![0-9]+]], [[META9:![0-9]+]], [[META10:![0-9]+]]}
+; UNROLLED: [[META8]] = !{!"llvm.loop.isvectorized", i32 1}
+; UNROLLED: [[META9]] = !{!"llvm.loop.vectorize.body", i32 1}
+; UNROLLED: [[META10]] = !{!"llvm.loop.unroll.runtime.disable"}
+;.
+; NONE: [[META2:![0-9]+]] = distinct !DICompileUnit(language: DW_LANG_C89, file: [[META3:![0-9]+]], isOptimized: false, runtimeVersion: 0, emissionKind: NoDebug)
+; NONE: [[META3]] = !DIFile(filename: "{{.*}}vectorization-remarks.c", directory: {{.*}})
+; NONE: [[DBG4]] = distinct !DISubprogram(name: "foo", scope: [[META3]], file: [[META3]], line: 5, type: [[META5:![0-9]+]], scopeLine: 6, virtualIndex: 6, flags: DIFlagPrototyped, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META2]], retainedNodes: [[META6:![0-9]+]])
+; NONE: [[META5]] = !DISubroutineType(types: [[META6]])
+; NONE: [[META6]] = !{}
+; NONE: [[LOOP7]] = distinct !{[[LOOP7]], [[META8:![0-9]+]]}
+; NONE: [[META8]] = !DILocation(line: 17, column: 8, scope: [[META9:![0-9]+]])
+; NONE: [[META9]] = distinct !DILexicalBlock(scope: [[META10:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; NONE: [[META10]] = distinct !DILexicalBlock(scope: [[META11:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; NONE: [[META11]] = distinct !DILexicalBlock(scope: [[DBG4]], file: [[META3]], line: 17, column: 3)
+;.
diff --git a/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks.ll b/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks.ll
index 41ad9ec20cc3d..398f2913bce75 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks.ll
@@ -1,14 +1,133 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
; RUN: opt < %s -passes=loop-vectorize -mtriple=x86_64-unknown-linux -S -pass-remarks='loop-vectorize' 2>&1 | FileCheck -check-prefix=VECTORIZED %s
; RUN: opt < %s -passes=loop-vectorize -force-vector-width=1 -force-vector-interleave=4 -mtriple=x86_64-unknown-linux -S -pass-remarks='loop-vectorize' 2>&1 | FileCheck -check-prefix=UNROLLED %s
; RUN: opt < %s -passes=loop-vectorize -force-vector-width=1 -force-vector-interleave=1 -mtriple=x86_64-unknown-linux -S -pass-remarks-analysis='loop-vectorize' 2>&1 | FileCheck -check-prefix=NONE %s
-; VECTORIZED: remark: vectorization-remarks.c:17:8: vectorized loop (vectorization width: 4, interleaved count: 2)
+; VECTORIZED: remark: vectorization-remarks.c:17:8: vectorized loop (vectorization width: 16, interleaved count: 1)
; UNROLLED: remark: vectorization-remarks.c:17:8: interleaved loop (interleaved count: 4)
; NONE: remark: vectorization-remarks.c:17:8: loop not vectorized: vectorization and interleaving are explicitly disabled, or the loop has already been vectorized
target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
define i32 @foo(i32 %n) #0 !dbg !4 {
+; VECTORIZED-LABEL: define i32 @foo(
+; VECTORIZED-SAME: i32 [[N:%.*]]) !dbg [[DBG4:![0-9]+]] {
+; VECTORIZED-NEXT: [[ENTRY:.*:]]
+; VECTORIZED-NEXT: [[DIFF:%.*]] = alloca i32, align 4
+; VECTORIZED-NEXT: [[CB:%.*]] = alloca [16 x i8], align 16
+; VECTORIZED-NEXT: [[CC:%.*]] = alloca [16 x i8], align 16
+; VECTORIZED-NEXT: store i32 0, ptr [[DIFF]], align 4, !dbg [[DBG7:![0-9]+]]
+; VECTORIZED-NEXT: br label %[[VECTOR_PH:.*]], !dbg [[DBG8:![0-9]+]]
+; VECTORIZED: [[VECTOR_PH]]:
+; VECTORIZED-NEXT: br label %[[VECTOR_BODY:.*]], !dbg [[DBG8]]
+; VECTORIZED: [[VECTOR_BODY]]:
+; VECTORIZED-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[CB]], align 1, !dbg [[DBG12:![0-9]+]]
+; VECTORIZED-NEXT: [[TMP0:%.*]] = sext <16 x i8> [[WIDE_LOAD]] to <16 x i32>, !dbg [[DBG12]]
+; VECTORIZED-NEXT: [[WIDE_LOAD1:%.*]] = load <16 x i8>, ptr [[CC]], align 1, !dbg [[DBG12]]
+; VECTORIZED-NEXT: [[TMP1:%.*]] = sext <16 x i8> [[WIDE_LOAD1]] to <16 x i32>, !dbg [[DBG12]]
+; VECTORIZED-NEXT: [[TMP2:%.*]] = sub <16 x i32> [[TMP0]], [[TMP1]], !dbg [[DBG12]]
+; VECTORIZED-NEXT: [[TMP3:%.*]] = add <16 x i32> [[TMP2]], zeroinitializer, !dbg [[DBG12]]
+; VECTORIZED-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; VECTORIZED: [[MIDDLE_BLOCK]]:
+; VECTORIZED-NEXT: [[TMP4:%.*]] = call i32 @llvm.vector.reduce.add.v16i32(<16 x i32> [[TMP3]]), !dbg [[DBG8]]
+; VECTORIZED-NEXT: br label %[[FOR_END:.*]], !dbg [[DBG12]]
+; VECTORIZED: [[FOR_END]]:
+; VECTORIZED-NEXT: store i32 [[TMP4]], ptr [[DIFF]], align 4, !dbg [[DBG12]]
+; VECTORIZED-NEXT: call void @ibar(ptr [[DIFF]]), !dbg [[DBG14:![0-9]+]]
+; VECTORIZED-NEXT: ret i32 0, !dbg [[DBG15:![0-9]+]]
+;
+; UNROLLED-LABEL: define i32 @foo(
+; UNROLLED-SAME: i32 [[N:%.*]]) !dbg [[DBG4:![0-9]+]] {
+; UNROLLED-NEXT: [[ENTRY:.*:]]
+; UNROLLED-NEXT: [[DIFF:%.*]] = alloca i32, align 4
+; UNROLLED-NEXT: [[CB:%.*]] = alloca [16 x i8], align 16
+; UNROLLED-NEXT: [[CC:%.*]] = alloca [16 x i8], align 16
+; UNROLLED-NEXT: store i32 0, ptr [[DIFF]], align 4, !dbg [[DBG7:![0-9]+]]
+; UNROLLED-NEXT: br label %[[VECTOR_PH:.*]], !dbg [[DBG8:![0-9]+]]
+; UNROLLED: [[VECTOR_PH]]:
+; UNROLLED-NEXT: br label %[[VECTOR_BODY:.*]], !dbg [[DBG8]]
+; UNROLLED: [[VECTOR_BODY]]:
+; UNROLLED-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ], !dbg [[DBG8]]
+; UNROLLED-NEXT: [[VEC_PHI:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP31:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[VEC_PHI1:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP32:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[VEC_PHI2:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP33:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[VEC_PHI3:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP34:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[TMP0:%.*]] = add i64 [[INDEX]], 1
+; UNROLLED-NEXT: [[TMP1:%.*]] = add i64 [[INDEX]], 2
+; UNROLLED-NEXT: [[TMP2:%.*]] = add i64 [[INDEX]], 3
+; UNROLLED-NEXT: [[TMP3:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[INDEX]], !dbg [[DBG12:![0-9]+]]
+; UNROLLED-NEXT: [[TMP4:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[TMP0]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP5:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[TMP1]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP6:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[TMP2]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP7:%.*]] = load i8, ptr [[TMP3]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP8:%.*]] = load i8, ptr [[TMP4]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP9:%.*]] = load i8, ptr [[TMP5]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP10:%.*]] = load i8, ptr [[TMP6]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP11:%.*]] = sext i8 [[TMP7]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP12:%.*]] = sext i8 [[TMP8]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP13:%.*]] = sext i8 [[TMP9]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP14:%.*]] = sext i8 [[TMP10]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP15:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[INDEX]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP16:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[TMP0]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP17:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[TMP1]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP18:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[TMP2]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP19:%.*]] = load i8, ptr [[TMP15]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP20:%.*]] = load i8, ptr [[TMP16]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP21:%.*]] = load i8, ptr [[TMP17]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP22:%.*]] = load i8, ptr [[TMP18]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP23:%.*]] = sext i8 [[TMP19]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP24:%.*]] = sext i8 [[TMP20]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP25:%.*]] = sext i8 [[TMP21]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP26:%.*]] = sext i8 [[TMP22]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP27:%.*]] = sub i32 [[TMP11]], [[TMP23]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP28:%.*]] = sub i32 [[TMP12]], [[TMP24]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP29:%.*]] = sub i32 [[TMP13]], [[TMP25]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP30:%.*]] = sub i32 [[TMP14]], [[TMP26]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP31]] = add i32 [[TMP27]], [[VEC_PHI]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP32]] = add i32 [[TMP28]], [[VEC_PHI1]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP33]] = add i32 [[TMP29]], [[VEC_PHI2]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP34]] = add i32 [[TMP30]], [[VEC_PHI3]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4, !dbg [[DBG8]]
+; UNROLLED-NEXT: [[TMP35:%.*]] = icmp eq i64 [[INDEX_NEXT]], 16, !dbg [[DBG8]]
+; UNROLLED-NEXT: br i1 [[TMP35]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !dbg [[DBG8]], !llvm.loop [[LOOP14:![0-9]+]]
+; UNROLLED: [[MIDDLE_BLOCK]]:
+; UNROLLED-NEXT: [[BIN_RDX:%.*]] = add i32 [[TMP32]], [[TMP31]], !dbg [[DBG8]]
+; UNROLLED-NEXT: [[BIN_RDX4:%.*]] = add i32 [[TMP33]], [[BIN_RDX]], !dbg [[DBG8]]
+; UNROLLED-NEXT: [[BIN_RDX5:%.*]] = add i32 [[TMP34]], [[BIN_RDX4]], !dbg [[DBG8]]
+; UNROLLED-NEXT: br label %[[FOR_END:.*]], !dbg [[DBG8]]
+; UNROLLED: [[FOR_END]]:
+; UNROLLED-NEXT: store i32 [[BIN_RDX5]], ptr [[DIFF]], align 4, !dbg [[DBG12]]
+; UNROLLED-NEXT: call void @ibar(ptr [[DIFF]]), !dbg [[DBG18:![0-9]+]]
+; UNROLLED-NEXT: ret i32 0, !dbg [[DBG19:![0-9]+]]
+;
+; NONE-LABEL: define i32 @foo(
+; NONE-SAME: i32 [[N:%.*]]) !dbg [[DBG4:![0-9]+]] {
+; NONE-NEXT: [[ENTRY:.*]]:
+; NONE-NEXT: [[DIFF:%.*]] = alloca i32, align 4
+; NONE-NEXT: [[CB:%.*]] = alloca [16 x i8], align 16
+; NONE-NEXT: [[CC:%.*]] = alloca [16 x i8], align 16
+; NONE-NEXT: store i32 0, ptr [[DIFF]], align 4, !dbg [[DBG7:![0-9]+]]
+; NONE-NEXT: br label %[[FOR_BODY:.*]], !dbg [[DBG8:![0-9]+]]
+; NONE: [[FOR_BODY]]:
+; NONE-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDVARS_IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; NONE-NEXT: [[ADD8:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ADD:%.*]], %[[FOR_BODY]] ], !dbg [[DBG12:![0-9]+]]
+; NONE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[INDVARS_IV]], !dbg [[DBG12]]
+; NONE-NEXT: [[TMP0:%.*]] = load i8, ptr [[ARRAYIDX]], align 1, !dbg [[DBG12]]
+; NONE-NEXT: [[CONV:%.*]] = sext i8 [[TMP0]] to i32, !dbg [[DBG12]]
+; NONE-NEXT: [[ARRAYIDX2:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[INDVARS_IV]], !dbg [[DBG12]]
+; NONE-NEXT: [[TMP1:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1, !dbg [[DBG12]]
+; NONE-NEXT: [[CONV3:%.*]] = sext i8 [[TMP1]] to i32, !dbg [[DBG12]]
+; NONE-NEXT: [[SUB:%.*]] = sub i32 [[CONV]], [[CONV3]], !dbg [[DBG12]]
+; NONE-NEXT: [[ADD]] = add nsw i32 [[SUB]], [[ADD8]], !dbg [[DBG12]]
+; NONE-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1, !dbg [[DBG8]]
+; NONE-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], 16, !dbg [[DBG8]]
+; NONE-NEXT: br i1 [[EXITCOND]], label %[[FOR_END:.*]], label %[[FOR_BODY]], !dbg [[DBG8]]
+; NONE: [[FOR_END]]:
+; NONE-NEXT: [[ADD_LCSSA:%.*]] = phi i32 [ [[ADD]], %[[FOR_BODY]] ], !dbg [[DBG12]]
+; NONE-NEXT: store i32 [[ADD_LCSSA]], ptr [[DIFF]], align 4, !dbg [[DBG12]]
+; NONE-NEXT: call void @ibar(ptr [[DIFF]]), !dbg [[DBG14:![0-9]+]]
+; NONE-NEXT: ret i32 0, !dbg [[DBG15:![0-9]+]]
+;
entry:
%diff = alloca i32, align 4
%cb = alloca [16 x i8], align 16
@@ -60,3 +179,53 @@ declare void @ibar(ptr)
!22 = !DILocation(line: 20, column: 3, scope: !4)
!23 = !DILocation(line: 21, column: 3, scope: !4)
!24 = distinct !DICompileUnit(language: DW_LANG_C89, file: !1, emissionKind: NoDebug)
+;.
+; VECTORIZED: [[META2:![0-9]+]] = distinct !DICompileUnit(language: DW_LANG_C89, file: [[META3:![0-9]+]], isOptimized: false, runtimeVersion: 0, emissionKind: NoDebug)
+; VECTORIZED: [[META3]] = !DIFile(filename: "{{.*}}vectorization-remarks.c", directory: {{.*}})
+; VECTORIZED: [[DBG4]] = distinct !DISubprogram(name: "foo", scope: [[META3]], file: [[META3]], line: 5, type: [[META5:![0-9]+]], scopeLine: 6, virtualIndex: 6, flags: DIFlagPrototyped, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META2]], retainedNodes: [[META6:![0-9]+]])
+; VECTORIZED: [[META5]] = !DISubroutineType(types: [[META6]])
+; VECTORIZED: [[META6]] = !{}
+; VECTORIZED: [[DBG7]] = !DILocation(line: 8, column: 3, scope: [[DBG4]])
+; VECTORIZED: [[DBG8]] = !DILocation(line: 17, column: 8, scope: [[META9:![0-9]+]])
+; VECTORIZED: [[META9]] = distinct !DILexicalBlock(scope: [[META10:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; VECTORIZED: [[META10]] = distinct !DILexicalBlock(scope: [[META11:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; VECTORIZED: [[META11]] = distinct !DILexicalBlock(scope: [[DBG4]], file: [[META3]], line: 17, column: 3)
+; VECTORIZED: [[DBG12]] = !DILocation(line: 18, column: 5, scope: [[META13:![0-9]+]])
+; VECTORIZED: [[META13]] = distinct !DILexicalBlock(scope: [[META11]], file: [[META3]], line: 17, column: 27)
+; VECTORIZED: [[DBG14]] = !DILocation(line: 20, column: 3, scope: [[DBG4]])
+; VECTORIZED: [[DBG15]] = !DILocation(line: 21, column: 3, scope: [[DBG4]])
+;.
+; UNROLLED: [[META2:![0-9]+]] = distinct !DICompileUnit(language: DW_LANG_C89, file: [[META3:![0-9]+]], isOptimized: false, runtimeVersion: 0, emissionKind: NoDebug)
+; UNROLLED: [[META3]] = !DIFile(filename: "{{.*}}vectorization-remarks.c", directory: {{.*}})
+; UNROLLED: [[DBG4]] = distinct !DISubprogram(name: "foo", scope: [[META3]], file: [[META3]], line: 5, type: [[META5:![0-9]+]], scopeLine: 6, virtualIndex: 6, flags: DIFlagPrototyped, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META2]], retainedNodes: [[META6:![0-9]+]])
+; UNROLLED: [[META5]] = !DISubroutineType(types: [[META6]])
+; UNROLLED: [[META6]] = !{}
+; UNROLLED: [[DBG7]] = !DILocation(line: 8, column: 3, scope: [[DBG4]])
+; UNROLLED: [[DBG8]] = !DILocation(line: 17, column: 8, scope: [[META9:![0-9]+]])
+; UNROLLED: [[META9]] = distinct !DILexicalBlock(scope: [[META10:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; UNROLLED: [[META10]] = distinct !DILexicalBlock(scope: [[META11:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; UNROLLED: [[META11]] = distinct !DILexicalBlock(scope: [[DBG4]], file: [[META3]], line: 17, column: 3)
+; UNROLLED: [[DBG12]] = !DILocation(line: 18, column: 5, scope: [[META13:![0-9]+]])
+; UNROLLED: [[META13]] = distinct !DILexicalBlock(scope: [[META11]], file: [[META3]], line: 17, column: 27)
+; UNROLLED: [[LOOP14]] = distinct !{[[LOOP14]], [[META15:![0-9]+]], [[META16:![0-9]+]], [[META17:![0-9]+]]}
+; UNROLLED: [[META15]] = !{!"llvm.loop.isvectorized", i32 1}
+; UNROLLED: [[META16]] = !{!"llvm.loop.vectorize.body", i32 1}
+; UNROLLED: [[META17]] = !{!"llvm.loop.unroll.runtime.disable"}
+; UNROLLED: [[DBG18]] = !DILocation(line: 20, column: 3, scope: [[DBG4]])
+; UNROLLED: [[DBG19]] = !DILocation(line: 21, column: 3, scope: [[DBG4]])
+;.
+; NONE: [[META2:![0-9]+]] = distinct !DICompileUnit(language: DW_LANG_C89, file: [[META3:![0-9]+]], isOptimized: false, runtimeVersion: 0, emissionKind: NoDebug)
+; NONE: [[META3]] = !DIFile(filename: "{{.*}}vectorization-remarks.c", directory: {{.*}})
+; NONE: [[DBG4]] = distinct !DISubprogram(name: "foo", scope: [[META3]], file: [[META3]], line: 5, type: [[META5:![0-9]+]], scopeLine: 6, virtualIndex: 6, flags: DIFlagPrototyped, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META2]], retainedNodes: [[META6:![0-9]+]])
+; NONE: [[META5]] = !DISubroutineType(types: [[META6]])
+; NONE: [[META6]] = !{}
+; NONE: [[DBG7]] = !DILocation(line: 8, column: 3, scope: [[DBG4]])
+; NONE: [[DBG8]] = !DILocation(line: 17, column: 8, scope: [[META9:![0-9]+]])
+; NONE: [[META9]] = distinct !DILexicalBlock(scope: [[META10:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; NONE: [[META10]] = distinct !DILexicalBlock(scope: [[META11:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; NONE: [[META11]] = distinct !DILexicalBlock(scope: [[DBG4]], file: [[META3]], line: 17, column: 3)
+; NONE: [[DBG12]] = !DILocation(line: 18, column: 5, scope: [[META13:![0-9]+]])
+; NONE: [[META13]] = distinct !DILexicalBlock(scope: [[META11]], file: [[META3]], line: 17, column: 27)
+; NONE: [[DBG14]] = !DILocation(line: 20, column: 3, scope: [[DBG4]])
+; NONE: [[DBG15]] = !DILocation(line: 21, column: 3, scope: [[DBG4]])
+;.
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/pixel-splat.ll b/llvm/test/Transforms/PhaseOrdering/X86/pixel-splat.ll
index 975de14a95fc4..d168f04485caf 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/pixel-splat.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/pixel-splat.ll
@@ -23,37 +23,56 @@ define void @loop_or(ptr noalias %pIn, ptr noalias %pOut, i32 %s) {
; CHECK-NEXT: entry:
; CHECK-NEXT: [[CMP1:%.*]] = icmp sgt i32 [[S:%.*]], 0
; CHECK-NEXT: br i1 [[CMP1]], label [[FOR_BODY_PREHEADER:%.*]], label [[FOR_END:%.*]]
-; CHECK: for.body.preheader:
+; CHECK: iter.check:
; CHECK-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext nneg i32 [[S]] to i64
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[S]], 8
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[S]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[FOR_BODY_PREHEADER5:%.*]], label [[VECTOR_PH:%.*]]
+; CHECK: vector.main.loop.iter.check:
+; CHECK-NEXT: [[MIN_ITERS_CHECK4:%.*]] = icmp ult i32 [[S]], 16
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK4]], label [[VEC_EPILOG_PH:%.*]], label [[VECTOR_PH1:%.*]]
; CHECK: vector.ph:
-; CHECK-NEXT: [[N_VEC:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 2147483640
+; CHECK-NEXT: [[TMP8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 12
+; CHECK-NEXT: [[N_VEC1:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 2147483632
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH1]] ], [ [[INDEX_NEXT1:%.*]], [[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[PIN:%.*]], i64 [[INDEX]]
-; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP0]], i64 4
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP2:%.*]] = zext <16 x i8> [[WIDE_LOAD]] to <16 x i32>
+; CHECK-NEXT: [[TMP12:%.*]] = mul nuw nsw <16 x i32> [[TMP2]], splat (i32 65793)
+; CHECK-NEXT: [[TMP4:%.*]] = or disjoint <16 x i32> [[TMP12]], splat (i32 -16777216)
+; CHECK-NEXT: [[TMP13:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[POUT:%.*]], i64 [[INDEX]]
+; CHECK-NEXT: store <16 x i32> [[TMP4]], ptr [[TMP13]], align 4
+; CHECK-NEXT: [[INDEX_NEXT1]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT1]], [[N_VEC1]]
+; CHECK-NEXT: br i1 [[TMP6]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[CMP_N1:%.*]] = icmp eq i64 [[N_VEC1]], [[WIDE_TRIP_COUNT]]
+; CHECK-NEXT: br i1 [[CMP_N1]], label [[FOR_END]], label [[VEC_EPILOG_ITER_CHECK:%.*]]
+; CHECK: vec.epilog.iter.check:
+; CHECK-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp eq i64 [[TMP8]], 0
+; CHECK-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label [[FOR_BODY_PREHEADER5]], label [[VEC_EPILOG_PH]], !prof [[PROF3:![0-9]+]]
+; CHECK: vec.epilog.ph:
+; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC1]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[VECTOR_PH]] ]
+; CHECK-NEXT: [[N_VEC:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 2147483644
+; CHECK-NEXT: br label [[VEC_EPILOG_VECTOR_BODY:%.*]]
+; CHECK: vec.epilog.vector.body:
+; CHECK-NEXT: [[INDEX6:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], [[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[PIN]], i64 [[INDEX6]]
; CHECK-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
-; CHECK-NEXT: [[TMP2:%.*]] = zext <4 x i8> [[WIDE_LOAD]] to <4 x i32>
; CHECK-NEXT: [[TMP3:%.*]] = zext <4 x i8> [[WIDE_LOAD4]] to <4 x i32>
-; CHECK-NEXT: [[TMP4:%.*]] = mul nuw nsw <4 x i32> [[TMP2]], splat (i32 65793)
; CHECK-NEXT: [[TMP5:%.*]] = mul nuw nsw <4 x i32> [[TMP3]], splat (i32 65793)
-; CHECK-NEXT: [[TMP6:%.*]] = or disjoint <4 x i32> [[TMP4]], splat (i32 -16777216)
; CHECK-NEXT: [[TMP7:%.*]] = or disjoint <4 x i32> [[TMP5]], splat (i32 -16777216)
-; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[POUT:%.*]], i64 [[INDEX]]
-; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP8]], i64 16
-; CHECK-NEXT: store <4 x i32> [[TMP6]], ptr [[TMP8]], align 4
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[POUT]], i64 [[INDEX6]]
; CHECK-NEXT: store <4 x i32> [[TMP7]], ptr [[TMP9]], align 4
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX6]], 4
; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT: br i1 [[TMP10]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK: middle.block:
+; CHECK-NEXT: br i1 [[TMP10]], label [[VEC_EPILOG_MIDDLE_BLOCK:%.*]], label [[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: vec.epilog.middle.block:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N_VEC]], [[WIDE_TRIP_COUNT]]
; CHECK-NEXT: br i1 [[CMP_N]], label [[FOR_END]], label [[FOR_BODY_PREHEADER5]]
-; CHECK: for.body.preheader5:
-; CHECK-NEXT: [[INDVARS_IV_PH:%.*]] = phi i64 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[N_VEC]], [[MIDDLE_BLOCK]] ]
+; CHECK: for.body.preheader:
+; CHECK-NEXT: [[INDVARS_IV_PH:%.*]] = phi i64 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[N_VEC1]], [[VEC_EPILOG_ITER_CHECK]] ], [ [[N_VEC]], [[VEC_EPILOG_MIDDLE_BLOCK]] ]
; CHECK-NEXT: br label [[FOR_BODY:%.*]]
; CHECK: for.body:
; CHECK-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY]] ], [ [[INDVARS_IV_PH]], [[FOR_BODY_PREHEADER5]] ]
@@ -66,7 +85,7 @@ define void @loop_or(ptr noalias %pIn, ptr noalias %pOut, i32 %s) {
; CHECK-NEXT: store i32 [[OR3]], ptr [[ARRAYIDX5]], align 4
; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1
; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], [[WIDE_TRIP_COUNT]]
-; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label [[FOR_END]], label [[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label [[FOR_END]], label [[FOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: for.end:
; CHECK-NEXT: ret void
;
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/preserve-access-group.ll b/llvm/test/Transforms/PhaseOrdering/X86/preserve-access-group.ll
index a66528f1f12b4..9ca53b494de2d 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/preserve-access-group.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/preserve-access-group.ll
@@ -15,27 +15,27 @@ define void @test(i32 noundef %nface, i32 noundef %ncell, ptr noalias noundef %f
; CHECK: [[FOR_BODY_PREHEADER]]:
; CHECK-NEXT: [[TMP0:%.*]] = zext nneg i32 [[NFACE]] to i64
; CHECK-NEXT: [[INVARIANT_GEP:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[FACE_CELL]], i64 [[TMP0]]
-; CHECK-NEXT: [[TMP1:%.*]] = icmp ult i32 [[NFACE]], 4
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ult i32 [[NFACE]], 8
; CHECK-NEXT: br i1 [[TMP1]], label %[[FOR_BODY_PREHEADER14:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[UNROLL_ITER:%.*]] = and i64 [[TMP0]], 2147483644
+; CHECK-NEXT: [[UNROLL_ITER:%.*]] = and i64 [[TMP0]], 2147483640
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDVARS_IV_EPIL:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[FACE_CELL]], i64 [[INDVARS_IV_EPIL]]
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP10]], align 4, !tbaa [[INT_TBAA0:![0-9]+]], !llvm.access.group [[ACC_GRP4:![0-9]+]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP10]], align 4, !tbaa [[INT_TBAA0:![0-9]+]], !llvm.access.group [[ACC_GRP4:![0-9]+]]
; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[INVARIANT_GEP]], i64 [[INDVARS_IV_EPIL]]
-; CHECK-NEXT: [[WIDE_LOAD12:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !tbaa [[INT_TBAA0]], !llvm.access.group [[ACC_GRP4]]
-; CHECK-NEXT: [[TMP3:%.*]] = sext <4 x i32> [[WIDE_LOAD]] to <4 x i64>
-; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds [8 x i8], ptr [[Y]], <4 x i64> [[TMP3]]
-; CHECK-NEXT: [[TMP5:%.*]] = sext <4 x i32> [[WIDE_LOAD12]] to <4 x i64>
-; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds [8 x i8], ptr [[X]], <4 x i64> [[TMP5]]
-; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = tail call <4 x double> @llvm.masked.gather.v4f64.v4p0(<4 x ptr> align 8 [[TMP4]], <4 x i1> splat (i1 true), <4 x double> poison), !tbaa [[DOUBLE_TBAA5:![0-9]+]], !llvm.access.group [[ACC_GRP4]]
-; CHECK-NEXT: [[WIDE_MASKED_GATHER13:%.*]] = tail call <4 x double> @llvm.masked.gather.v4f64.v4p0(<4 x ptr> align 8 [[TMP6]], <4 x i1> splat (i1 true), <4 x double> poison), !tbaa [[DOUBLE_TBAA5]], !llvm.access.group [[ACC_GRP4]]
-; CHECK-NEXT: [[TMP7:%.*]] = fcmp fast olt <4 x double> [[WIDE_MASKED_GATHER]], [[WIDE_MASKED_GATHER13]]
-; CHECK-NEXT: [[TMP8:%.*]] = select <4 x i1> [[TMP7]], <4 x double> [[WIDE_MASKED_GATHER13]], <4 x double> [[WIDE_MASKED_GATHER]]
-; CHECK-NEXT: tail call void @llvm.masked.scatter.v4f64.v4p0(<4 x double> [[TMP8]], <4 x ptr> align 8 [[TMP4]], <4 x i1> splat (i1 true)), !tbaa [[DOUBLE_TBAA5]], !llvm.access.group [[ACC_GRP4]]
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDVARS_IV_EPIL]], 4
+; CHECK-NEXT: [[WIDE_LOAD12:%.*]] = load <8 x i32>, ptr [[TMP2]], align 4, !tbaa [[INT_TBAA0]], !llvm.access.group [[ACC_GRP4]]
+; CHECK-NEXT: [[TMP3:%.*]] = sext <8 x i32> [[WIDE_LOAD]] to <8 x i64>
+; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds [8 x i8], ptr [[Y]], <8 x i64> [[TMP3]]
+; CHECK-NEXT: [[TMP4:%.*]] = sext <8 x i32> [[WIDE_LOAD12]] to <8 x i64>
+; CHECK-NEXT: [[WIDE_GEP13:%.*]] = getelementptr inbounds [8 x i8], ptr [[X]], <8 x i64> [[TMP4]]
+; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = tail call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 8 [[WIDE_GEP]], <8 x i1> splat (i1 true), <8 x double> poison), !tbaa [[DOUBLE_TBAA5:![0-9]+]], !llvm.access.group [[ACC_GRP4]]
+; CHECK-NEXT: [[WIDE_MASKED_GATHER14:%.*]] = tail call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 8 [[WIDE_GEP13]], <8 x i1> splat (i1 true), <8 x double> poison), !tbaa [[DOUBLE_TBAA5]], !llvm.access.group [[ACC_GRP4]]
+; CHECK-NEXT: [[TMP5:%.*]] = fcmp fast olt <8 x double> [[WIDE_MASKED_GATHER]], [[WIDE_MASKED_GATHER14]]
+; CHECK-NEXT: [[TMP6:%.*]] = select <8 x i1> [[TMP5]], <8 x double> [[WIDE_MASKED_GATHER14]], <8 x double> [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT: tail call void @llvm.masked.scatter.v8f64.v8p0(<8 x double> [[TMP6]], <8 x ptr> align 8 [[WIDE_GEP]], <8 x i1> splat (i1 true)), !tbaa [[DOUBLE_TBAA5]], !llvm.access.group [[ACC_GRP4]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDVARS_IV_EPIL]], 8
; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[UNROLL_ITER]]
; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-known-first-value.ll b/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-known-first-value.ll
index 149dac30062cf..5de11ef4808ff 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-known-first-value.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-known-first-value.ll
@@ -11,25 +11,261 @@ define i16 @test(ptr %ptr) {
; CHECK-NEXT: entry:
; CHECK-NEXT: [[FIRST:%.*]] = load i8, ptr [[PTR:%.*]], align 1
; CHECK-NEXT: tail call void @use(i8 [[FIRST]]) #[[ATTR2:[0-9]+]]
-; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
-; CHECK: vector.body:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <8 x i16> [ zeroinitializer, [[ENTRY]] ], [ [[TMP4:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI1:%.*]] = phi <8 x i16> [ zeroinitializer, [[ENTRY]] ], [ [[TMP5:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 [[INDEX]]
-; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP0]], i64 8
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i8>, ptr [[TMP0]], align 1
-; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <8 x i8>, ptr [[TMP1]], align 1
-; CHECK-NEXT: [[TMP2:%.*]] = zext <8 x i8> [[WIDE_LOAD]] to <8 x i16>
-; CHECK-NEXT: [[TMP3:%.*]] = zext <8 x i8> [[WIDE_LOAD2]] to <8 x i16>
-; CHECK-NEXT: [[TMP4]] = add <8 x i16> [[VEC_PHI]], [[TMP2]]
-; CHECK-NEXT: [[TMP5]] = add <8 x i16> [[VEC_PHI1]], [[TMP3]]
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
-; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
-; CHECK-NEXT: br i1 [[TMP6]], label [[EXIT:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK: exit:
-; CHECK-NEXT: [[BIN_RDX:%.*]] = add <8 x i16> [[TMP5]], [[TMP4]]
-; CHECK-NEXT: [[TMP7:%.*]] = tail call i16 @llvm.vector.reduce.add.v8i16(<8 x i16> [[BIN_RDX]])
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 16
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[PTR]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = zext <16 x i8> [[WIDE_LOAD]] to <16 x i16>
+; CHECK-NEXT: [[TMP2:%.*]] = zext <16 x i8> [[WIDE_LOAD2]] to <16 x i16>
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 32
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 48
+; CHECK-NEXT: [[WIDE_LOAD_1:%.*]] = load <16 x i8>, ptr [[TMP3]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_1:%.*]] = load <16 x i8>, ptr [[TMP4]], align 1
+; CHECK-NEXT: [[TMP5:%.*]] = zext <16 x i8> [[WIDE_LOAD_1]] to <16 x i16>
+; CHECK-NEXT: [[TMP6:%.*]] = zext <16 x i8> [[WIDE_LOAD2_1]] to <16 x i16>
+; CHECK-NEXT: [[TMP189:%.*]] = add nuw nsw <16 x i16> [[TMP1]], [[TMP5]]
+; CHECK-NEXT: [[TMP8:%.*]] = add nuw nsw <16 x i16> [[TMP2]], [[TMP6]]
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 64
+; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 80
+; CHECK-NEXT: [[WIDE_LOAD_2:%.*]] = load <16 x i8>, ptr [[TMP9]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_2:%.*]] = load <16 x i8>, ptr [[TMP10]], align 1
+; CHECK-NEXT: [[TMP11:%.*]] = zext <16 x i8> [[WIDE_LOAD_2]] to <16 x i16>
+; CHECK-NEXT: [[TMP12:%.*]] = zext <16 x i8> [[WIDE_LOAD2_2]] to <16 x i16>
+; CHECK-NEXT: [[TMP13:%.*]] = add nuw nsw <16 x i16> [[TMP189]], [[TMP11]]
+; CHECK-NEXT: [[TMP14:%.*]] = add nuw nsw <16 x i16> [[TMP8]], [[TMP12]]
+; CHECK-NEXT: [[TMP15:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 96
+; CHECK-NEXT: [[TMP16:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 112
+; CHECK-NEXT: [[WIDE_LOAD_3:%.*]] = load <16 x i8>, ptr [[TMP15]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_3:%.*]] = load <16 x i8>, ptr [[TMP16]], align 1
+; CHECK-NEXT: [[TMP17:%.*]] = zext <16 x i8> [[WIDE_LOAD_3]] to <16 x i16>
+; CHECK-NEXT: [[TMP18:%.*]] = zext <16 x i8> [[WIDE_LOAD2_3]] to <16 x i16>
+; CHECK-NEXT: [[TMP19:%.*]] = add nuw nsw <16 x i16> [[TMP13]], [[TMP17]]
+; CHECK-NEXT: [[TMP20:%.*]] = add nuw nsw <16 x i16> [[TMP14]], [[TMP18]]
+; CHECK-NEXT: [[TMP21:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 128
+; CHECK-NEXT: [[TMP22:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 144
+; CHECK-NEXT: [[WIDE_LOAD_4:%.*]] = load <16 x i8>, ptr [[TMP21]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_4:%.*]] = load <16 x i8>, ptr [[TMP22]], align 1
+; CHECK-NEXT: [[TMP23:%.*]] = zext <16 x i8> [[WIDE_LOAD_4]] to <16 x i16>
+; CHECK-NEXT: [[TMP24:%.*]] = zext <16 x i8> [[WIDE_LOAD2_4]] to <16 x i16>
+; CHECK-NEXT: [[TMP25:%.*]] = add nuw nsw <16 x i16> [[TMP19]], [[TMP23]]
+; CHECK-NEXT: [[TMP26:%.*]] = add nuw nsw <16 x i16> [[TMP20]], [[TMP24]]
+; CHECK-NEXT: [[TMP27:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 160
+; CHECK-NEXT: [[TMP28:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 176
+; CHECK-NEXT: [[WIDE_LOAD_5:%.*]] = load <16 x i8>, ptr [[TMP27]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_5:%.*]] = load <16 x i8>, ptr [[TMP28]], align 1
+; CHECK-NEXT: [[TMP29:%.*]] = zext <16 x i8> [[WIDE_LOAD_5]] to <16 x i16>
+; CHECK-NEXT: [[TMP30:%.*]] = zext <16 x i8> [[WIDE_LOAD2_5]] to <16 x i16>
+; CHECK-NEXT: [[TMP31:%.*]] = add nuw nsw <16 x i16> [[TMP25]], [[TMP29]]
+; CHECK-NEXT: [[TMP32:%.*]] = add nuw nsw <16 x i16> [[TMP26]], [[TMP30]]
+; CHECK-NEXT: [[TMP33:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 192
+; CHECK-NEXT: [[TMP34:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 208
+; CHECK-NEXT: [[WIDE_LOAD_6:%.*]] = load <16 x i8>, ptr [[TMP33]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_6:%.*]] = load <16 x i8>, ptr [[TMP34]], align 1
+; CHECK-NEXT: [[TMP35:%.*]] = zext <16 x i8> [[WIDE_LOAD_6]] to <16 x i16>
+; CHECK-NEXT: [[TMP36:%.*]] = zext <16 x i8> [[WIDE_LOAD2_6]] to <16 x i16>
+; CHECK-NEXT: [[TMP37:%.*]] = add nuw nsw <16 x i16> [[TMP31]], [[TMP35]]
+; CHECK-NEXT: [[TMP38:%.*]] = add nuw nsw <16 x i16> [[TMP32]], [[TMP36]]
+; CHECK-NEXT: [[TMP39:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 224
+; CHECK-NEXT: [[TMP40:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 240
+; CHECK-NEXT: [[WIDE_LOAD_7:%.*]] = load <16 x i8>, ptr [[TMP39]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_7:%.*]] = load <16 x i8>, ptr [[TMP40]], align 1
+; CHECK-NEXT: [[TMP41:%.*]] = zext <16 x i8> [[WIDE_LOAD_7]] to <16 x i16>
+; CHECK-NEXT: [[TMP42:%.*]] = zext <16 x i8> [[WIDE_LOAD2_7]] to <16 x i16>
+; CHECK-NEXT: [[TMP43:%.*]] = add <16 x i16> [[TMP37]], [[TMP41]]
+; CHECK-NEXT: [[TMP44:%.*]] = add <16 x i16> [[TMP38]], [[TMP42]]
+; CHECK-NEXT: [[TMP45:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 256
+; CHECK-NEXT: [[TMP46:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 272
+; CHECK-NEXT: [[WIDE_LOAD_8:%.*]] = load <16 x i8>, ptr [[TMP45]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_8:%.*]] = load <16 x i8>, ptr [[TMP46]], align 1
+; CHECK-NEXT: [[TMP47:%.*]] = zext <16 x i8> [[WIDE_LOAD_8]] to <16 x i16>
+; CHECK-NEXT: [[TMP48:%.*]] = zext <16 x i8> [[WIDE_LOAD2_8]] to <16 x i16>
+; CHECK-NEXT: [[TMP49:%.*]] = add <16 x i16> [[TMP43]], [[TMP47]]
+; CHECK-NEXT: [[TMP50:%.*]] = add <16 x i16> [[TMP44]], [[TMP48]]
+; CHECK-NEXT: [[TMP51:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 288
+; CHECK-NEXT: [[TMP52:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 304
+; CHECK-NEXT: [[WIDE_LOAD_9:%.*]] = load <16 x i8>, ptr [[TMP51]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_9:%.*]] = load <16 x i8>, ptr [[TMP52]], align 1
+; CHECK-NEXT: [[TMP53:%.*]] = zext <16 x i8> [[WIDE_LOAD_9]] to <16 x i16>
+; CHECK-NEXT: [[TMP54:%.*]] = zext <16 x i8> [[WIDE_LOAD2_9]] to <16 x i16>
+; CHECK-NEXT: [[TMP55:%.*]] = add <16 x i16> [[TMP49]], [[TMP53]]
+; CHECK-NEXT: [[TMP56:%.*]] = add <16 x i16> [[TMP50]], [[TMP54]]
+; CHECK-NEXT: [[TMP57:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 320
+; CHECK-NEXT: [[TMP58:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 336
+; CHECK-NEXT: [[WIDE_LOAD_10:%.*]] = load <16 x i8>, ptr [[TMP57]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_10:%.*]] = load <16 x i8>, ptr [[TMP58]], align 1
+; CHECK-NEXT: [[TMP59:%.*]] = zext <16 x i8> [[WIDE_LOAD_10]] to <16 x i16>
+; CHECK-NEXT: [[TMP60:%.*]] = zext <16 x i8> [[WIDE_LOAD2_10]] to <16 x i16>
+; CHECK-NEXT: [[TMP61:%.*]] = add <16 x i16> [[TMP55]], [[TMP59]]
+; CHECK-NEXT: [[TMP62:%.*]] = add <16 x i16> [[TMP56]], [[TMP60]]
+; CHECK-NEXT: [[TMP63:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 352
+; CHECK-NEXT: [[TMP64:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 368
+; CHECK-NEXT: [[WIDE_LOAD_11:%.*]] = load <16 x i8>, ptr [[TMP63]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_11:%.*]] = load <16 x i8>, ptr [[TMP64]], align 1
+; CHECK-NEXT: [[TMP65:%.*]] = zext <16 x i8> [[WIDE_LOAD_11]] to <16 x i16>
+; CHECK-NEXT: [[TMP66:%.*]] = zext <16 x i8> [[WIDE_LOAD2_11]] to <16 x i16>
+; CHECK-NEXT: [[TMP67:%.*]] = add <16 x i16> [[TMP61]], [[TMP65]]
+; CHECK-NEXT: [[TMP68:%.*]] = add <16 x i16> [[TMP62]], [[TMP66]]
+; CHECK-NEXT: [[TMP69:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 384
+; CHECK-NEXT: [[TMP70:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 400
+; CHECK-NEXT: [[WIDE_LOAD_12:%.*]] = load <16 x i8>, ptr [[TMP69]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_12:%.*]] = load <16 x i8>, ptr [[TMP70]], align 1
+; CHECK-NEXT: [[TMP71:%.*]] = zext <16 x i8> [[WIDE_LOAD_12]] to <16 x i16>
+; CHECK-NEXT: [[TMP72:%.*]] = zext <16 x i8> [[WIDE_LOAD2_12]] to <16 x i16>
+; CHECK-NEXT: [[TMP73:%.*]] = add <16 x i16> [[TMP67]], [[TMP71]]
+; CHECK-NEXT: [[TMP74:%.*]] = add <16 x i16> [[TMP68]], [[TMP72]]
+; CHECK-NEXT: [[TMP75:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 416
+; CHECK-NEXT: [[TMP76:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 432
+; CHECK-NEXT: [[WIDE_LOAD_13:%.*]] = load <16 x i8>, ptr [[TMP75]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_13:%.*]] = load <16 x i8>, ptr [[TMP76]], align 1
+; CHECK-NEXT: [[TMP77:%.*]] = zext <16 x i8> [[WIDE_LOAD_13]] to <16 x i16>
+; CHECK-NEXT: [[TMP78:%.*]] = zext <16 x i8> [[WIDE_LOAD2_13]] to <16 x i16>
+; CHECK-NEXT: [[TMP79:%.*]] = add <16 x i16> [[TMP73]], [[TMP77]]
+; CHECK-NEXT: [[TMP80:%.*]] = add <16 x i16> [[TMP74]], [[TMP78]]
+; CHECK-NEXT: [[TMP81:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 448
+; CHECK-NEXT: [[TMP82:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 464
+; CHECK-NEXT: [[WIDE_LOAD_14:%.*]] = load <16 x i8>, ptr [[TMP81]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_14:%.*]] = load <16 x i8>, ptr [[TMP82]], align 1
+; CHECK-NEXT: [[TMP83:%.*]] = zext <16 x i8> [[WIDE_LOAD_14]] to <16 x i16>
+; CHECK-NEXT: [[TMP84:%.*]] = zext <16 x i8> [[WIDE_LOAD2_14]] to <16 x i16>
+; CHECK-NEXT: [[TMP85:%.*]] = add <16 x i16> [[TMP79]], [[TMP83]]
+; CHECK-NEXT: [[TMP86:%.*]] = add <16 x i16> [[TMP80]], [[TMP84]]
+; CHECK-NEXT: [[TMP87:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 480
+; CHECK-NEXT: [[TMP88:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 496
+; CHECK-NEXT: [[WIDE_LOAD_15:%.*]] = load <16 x i8>, ptr [[TMP87]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_15:%.*]] = load <16 x i8>, ptr [[TMP88]], align 1
+; CHECK-NEXT: [[TMP89:%.*]] = zext <16 x i8> [[WIDE_LOAD_15]] to <16 x i16>
+; CHECK-NEXT: [[TMP90:%.*]] = zext <16 x i8> [[WIDE_LOAD2_15]] to <16 x i16>
+; CHECK-NEXT: [[TMP91:%.*]] = add <16 x i16> [[TMP85]], [[TMP89]]
+; CHECK-NEXT: [[TMP92:%.*]] = add <16 x i16> [[TMP86]], [[TMP90]]
+; CHECK-NEXT: [[TMP93:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 512
+; CHECK-NEXT: [[TMP94:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 528
+; CHECK-NEXT: [[WIDE_LOAD_16:%.*]] = load <16 x i8>, ptr [[TMP93]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_16:%.*]] = load <16 x i8>, ptr [[TMP94]], align 1
+; CHECK-NEXT: [[TMP95:%.*]] = zext <16 x i8> [[WIDE_LOAD_16]] to <16 x i16>
+; CHECK-NEXT: [[TMP96:%.*]] = zext <16 x i8> [[WIDE_LOAD2_16]] to <16 x i16>
+; CHECK-NEXT: [[TMP97:%.*]] = add <16 x i16> [[TMP91]], [[TMP95]]
+; CHECK-NEXT: [[TMP98:%.*]] = add <16 x i16> [[TMP92]], [[TMP96]]
+; CHECK-NEXT: [[TMP99:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 544
+; CHECK-NEXT: [[TMP100:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 560
+; CHECK-NEXT: [[WIDE_LOAD_17:%.*]] = load <16 x i8>, ptr [[TMP99]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_17:%.*]] = load <16 x i8>, ptr [[TMP100]], align 1
+; CHECK-NEXT: [[TMP101:%.*]] = zext <16 x i8> [[WIDE_LOAD_17]] to <16 x i16>
+; CHECK-NEXT: [[TMP102:%.*]] = zext <16 x i8> [[WIDE_LOAD2_17]] to <16 x i16>
+; CHECK-NEXT: [[TMP103:%.*]] = add <16 x i16> [[TMP97]], [[TMP101]]
+; CHECK-NEXT: [[TMP104:%.*]] = add <16 x i16> [[TMP98]], [[TMP102]]
+; CHECK-NEXT: [[TMP105:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 576
+; CHECK-NEXT: [[TMP106:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 592
+; CHECK-NEXT: [[WIDE_LOAD_18:%.*]] = load <16 x i8>, ptr [[TMP105]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_18:%.*]] = load <16 x i8>, ptr [[TMP106]], align 1
+; CHECK-NEXT: [[TMP107:%.*]] = zext <16 x i8> [[WIDE_LOAD_18]] to <16 x i16>
+; CHECK-NEXT: [[TMP108:%.*]] = zext <16 x i8> [[WIDE_LOAD2_18]] to <16 x i16>
+; CHECK-NEXT: [[TMP109:%.*]] = add <16 x i16> [[TMP103]], [[TMP107]]
+; CHECK-NEXT: [[TMP110:%.*]] = add <16 x i16> [[TMP104]], [[TMP108]]
+; CHECK-NEXT: [[TMP111:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 608
+; CHECK-NEXT: [[TMP112:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 624
+; CHECK-NEXT: [[WIDE_LOAD_19:%.*]] = load <16 x i8>, ptr [[TMP111]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_19:%.*]] = load <16 x i8>, ptr [[TMP112]], align 1
+; CHECK-NEXT: [[TMP113:%.*]] = zext <16 x i8> [[WIDE_LOAD_19]] to <16 x i16>
+; CHECK-NEXT: [[TMP114:%.*]] = zext <16 x i8> [[WIDE_LOAD2_19]] to <16 x i16>
+; CHECK-NEXT: [[TMP115:%.*]] = add <16 x i16> [[TMP109]], [[TMP113]]
+; CHECK-NEXT: [[TMP116:%.*]] = add <16 x i16> [[TMP110]], [[TMP114]]
+; CHECK-NEXT: [[TMP117:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 640
+; CHECK-NEXT: [[TMP118:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 656
+; CHECK-NEXT: [[WIDE_LOAD_20:%.*]] = load <16 x i8>, ptr [[TMP117]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_20:%.*]] = load <16 x i8>, ptr [[TMP118]], align 1
+; CHECK-NEXT: [[TMP119:%.*]] = zext <16 x i8> [[WIDE_LOAD_20]] to <16 x i16>
+; CHECK-NEXT: [[TMP120:%.*]] = zext <16 x i8> [[WIDE_LOAD2_20]] to <16 x i16>
+; CHECK-NEXT: [[TMP121:%.*]] = add <16 x i16> [[TMP115]], [[TMP119]]
+; CHECK-NEXT: [[TMP122:%.*]] = add <16 x i16> [[TMP116]], [[TMP120]]
+; CHECK-NEXT: [[TMP123:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 672
+; CHECK-NEXT: [[TMP124:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 688
+; CHECK-NEXT: [[WIDE_LOAD_21:%.*]] = load <16 x i8>, ptr [[TMP123]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_21:%.*]] = load <16 x i8>, ptr [[TMP124]], align 1
+; CHECK-NEXT: [[TMP125:%.*]] = zext <16 x i8> [[WIDE_LOAD_21]] to <16 x i16>
+; CHECK-NEXT: [[TMP126:%.*]] = zext <16 x i8> [[WIDE_LOAD2_21]] to <16 x i16>
+; CHECK-NEXT: [[TMP127:%.*]] = add <16 x i16> [[TMP121]], [[TMP125]]
+; CHECK-NEXT: [[TMP128:%.*]] = add <16 x i16> [[TMP122]], [[TMP126]]
+; CHECK-NEXT: [[TMP129:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 704
+; CHECK-NEXT: [[TMP130:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 720
+; CHECK-NEXT: [[WIDE_LOAD_22:%.*]] = load <16 x i8>, ptr [[TMP129]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_22:%.*]] = load <16 x i8>, ptr [[TMP130]], align 1
+; CHECK-NEXT: [[TMP131:%.*]] = zext <16 x i8> [[WIDE_LOAD_22]] to <16 x i16>
+; CHECK-NEXT: [[TMP132:%.*]] = zext <16 x i8> [[WIDE_LOAD2_22]] to <16 x i16>
+; CHECK-NEXT: [[TMP133:%.*]] = add <16 x i16> [[TMP127]], [[TMP131]]
+; CHECK-NEXT: [[TMP134:%.*]] = add <16 x i16> [[TMP128]], [[TMP132]]
+; CHECK-NEXT: [[TMP135:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 736
+; CHECK-NEXT: [[TMP136:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 752
+; CHECK-NEXT: [[WIDE_LOAD_23:%.*]] = load <16 x i8>, ptr [[TMP135]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_23:%.*]] = load <16 x i8>, ptr [[TMP136]], align 1
+; CHECK-NEXT: [[TMP137:%.*]] = zext <16 x i8> [[WIDE_LOAD_23]] to <16 x i16>
+; CHECK-NEXT: [[TMP138:%.*]] = zext <16 x i8> [[WIDE_LOAD2_23]] to <16 x i16>
+; CHECK-NEXT: [[TMP139:%.*]] = add <16 x i16> [[TMP133]], [[TMP137]]
+; CHECK-NEXT: [[TMP140:%.*]] = add <16 x i16> [[TMP134]], [[TMP138]]
+; CHECK-NEXT: [[TMP141:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 768
+; CHECK-NEXT: [[TMP142:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 784
+; CHECK-NEXT: [[WIDE_LOAD_24:%.*]] = load <16 x i8>, ptr [[TMP141]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_24:%.*]] = load <16 x i8>, ptr [[TMP142]], align 1
+; CHECK-NEXT: [[TMP143:%.*]] = zext <16 x i8> [[WIDE_LOAD_24]] to <16 x i16>
+; CHECK-NEXT: [[TMP144:%.*]] = zext <16 x i8> [[WIDE_LOAD2_24]] to <16 x i16>
+; CHECK-NEXT: [[TMP145:%.*]] = add <16 x i16> [[TMP139]], [[TMP143]]
+; CHECK-NEXT: [[TMP146:%.*]] = add <16 x i16> [[TMP140]], [[TMP144]]
+; CHECK-NEXT: [[TMP147:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 800
+; CHECK-NEXT: [[TMP148:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 816
+; CHECK-NEXT: [[WIDE_LOAD_25:%.*]] = load <16 x i8>, ptr [[TMP147]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_25:%.*]] = load <16 x i8>, ptr [[TMP148]], align 1
+; CHECK-NEXT: [[TMP149:%.*]] = zext <16 x i8> [[WIDE_LOAD_25]] to <16 x i16>
+; CHECK-NEXT: [[TMP150:%.*]] = zext <16 x i8> [[WIDE_LOAD2_25]] to <16 x i16>
+; CHECK-NEXT: [[TMP151:%.*]] = add <16 x i16> [[TMP145]], [[TMP149]]
+; CHECK-NEXT: [[TMP152:%.*]] = add <16 x i16> [[TMP146]], [[TMP150]]
+; CHECK-NEXT: [[TMP153:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 832
+; CHECK-NEXT: [[TMP154:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 848
+; CHECK-NEXT: [[WIDE_LOAD_26:%.*]] = load <16 x i8>, ptr [[TMP153]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_26:%.*]] = load <16 x i8>, ptr [[TMP154]], align 1
+; CHECK-NEXT: [[TMP155:%.*]] = zext <16 x i8> [[WIDE_LOAD_26]] to <16 x i16>
+; CHECK-NEXT: [[TMP156:%.*]] = zext <16 x i8> [[WIDE_LOAD2_26]] to <16 x i16>
+; CHECK-NEXT: [[TMP157:%.*]] = add <16 x i16> [[TMP151]], [[TMP155]]
+; CHECK-NEXT: [[TMP158:%.*]] = add <16 x i16> [[TMP152]], [[TMP156]]
+; CHECK-NEXT: [[TMP159:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 864
+; CHECK-NEXT: [[TMP160:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 880
+; CHECK-NEXT: [[WIDE_LOAD_27:%.*]] = load <16 x i8>, ptr [[TMP159]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_27:%.*]] = load <16 x i8>, ptr [[TMP160]], align 1
+; CHECK-NEXT: [[TMP161:%.*]] = zext <16 x i8> [[WIDE_LOAD_27]] to <16 x i16>
+; CHECK-NEXT: [[TMP162:%.*]] = zext <16 x i8> [[WIDE_LOAD2_27]] to <16 x i16>
+; CHECK-NEXT: [[TMP163:%.*]] = add <16 x i16> [[TMP157]], [[TMP161]]
+; CHECK-NEXT: [[TMP164:%.*]] = add <16 x i16> [[TMP158]], [[TMP162]]
+; CHECK-NEXT: [[TMP165:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 896
+; CHECK-NEXT: [[TMP166:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 912
+; CHECK-NEXT: [[WIDE_LOAD_28:%.*]] = load <16 x i8>, ptr [[TMP165]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_28:%.*]] = load <16 x i8>, ptr [[TMP166]], align 1
+; CHECK-NEXT: [[TMP167:%.*]] = zext <16 x i8> [[WIDE_LOAD_28]] to <16 x i16>
+; CHECK-NEXT: [[TMP168:%.*]] = zext <16 x i8> [[WIDE_LOAD2_28]] to <16 x i16>
+; CHECK-NEXT: [[TMP169:%.*]] = add <16 x i16> [[TMP163]], [[TMP167]]
+; CHECK-NEXT: [[TMP170:%.*]] = add <16 x i16> [[TMP164]], [[TMP168]]
+; CHECK-NEXT: [[TMP171:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 928
+; CHECK-NEXT: [[TMP172:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 944
+; CHECK-NEXT: [[WIDE_LOAD_29:%.*]] = load <16 x i8>, ptr [[TMP171]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_29:%.*]] = load <16 x i8>, ptr [[TMP172]], align 1
+; CHECK-NEXT: [[TMP173:%.*]] = zext <16 x i8> [[WIDE_LOAD_29]] to <16 x i16>
+; CHECK-NEXT: [[TMP174:%.*]] = zext <16 x i8> [[WIDE_LOAD2_29]] to <16 x i16>
+; CHECK-NEXT: [[TMP175:%.*]] = add <16 x i16> [[TMP169]], [[TMP173]]
+; CHECK-NEXT: [[TMP176:%.*]] = add <16 x i16> [[TMP170]], [[TMP174]]
+; CHECK-NEXT: [[TMP177:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 960
+; CHECK-NEXT: [[TMP178:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 976
+; CHECK-NEXT: [[WIDE_LOAD_30:%.*]] = load <16 x i8>, ptr [[TMP177]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_30:%.*]] = load <16 x i8>, ptr [[TMP178]], align 1
+; CHECK-NEXT: [[TMP179:%.*]] = zext <16 x i8> [[WIDE_LOAD_30]] to <16 x i16>
+; CHECK-NEXT: [[TMP180:%.*]] = zext <16 x i8> [[WIDE_LOAD2_30]] to <16 x i16>
+; CHECK-NEXT: [[TMP181:%.*]] = add <16 x i16> [[TMP175]], [[TMP179]]
+; CHECK-NEXT: [[TMP182:%.*]] = add <16 x i16> [[TMP176]], [[TMP180]]
+; CHECK-NEXT: [[TMP183:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 992
+; CHECK-NEXT: [[TMP184:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 1008
+; CHECK-NEXT: [[WIDE_LOAD_31:%.*]] = load <16 x i8>, ptr [[TMP183]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_31:%.*]] = load <16 x i8>, ptr [[TMP184]], align 1
+; CHECK-NEXT: [[TMP185:%.*]] = zext <16 x i8> [[WIDE_LOAD_31]] to <16 x i16>
+; CHECK-NEXT: [[TMP186:%.*]] = zext <16 x i8> [[WIDE_LOAD2_31]] to <16 x i16>
+; CHECK-NEXT: [[TMP187:%.*]] = add <16 x i16> [[TMP181]], [[TMP185]]
+; CHECK-NEXT: [[TMP188:%.*]] = add <16 x i16> [[TMP182]], [[TMP186]]
+; CHECK-NEXT: [[BIN_RDX:%.*]] = add <16 x i16> [[TMP188]], [[TMP187]]
+; CHECK-NEXT: [[TMP7:%.*]] = tail call i16 @llvm.vector.reduce.add.v16i16(<16 x i16> [[BIN_RDX]])
; CHECK-NEXT: ret i16 [[TMP7]]
;
entry:
>From 6a11d485d63a4d57ebdfbf00db052f24a2d5c16a Mon Sep 17 00:00:00 2001
From: Sumukh Bharadwaj <Sumukh.Bharadwaj at amd.com>
Date: Mon, 7 Sep 2026 14:39:01 +0530
Subject: [PATCH 2/2] [X86][CostModel] Price predicate mask-expansion fanout
for wide predicated ops
A masked memory op, gather/scatter or select produces its predicate as a
compact <N x i1>, but when the value/data type legalizes into more
register parts than that predicate, codegen must materialize a sub-mask
for every extra data part -- a kshiftr out of the k-register on AVX-512,
or a vector unpack/sign-extend on SSE/AVX. The per-part cost tables price
these ops linearly in data parts and miss the fanout, so per-lane cost
stays flat as the type widens. With MaximizeBandwidth sizing VF from the
narrowest type and the vectorizer tie-breaking toward the widest VF,
nothing penalizes over-widening and predicated loops blow up -- e.g. a
conditional store picking VF32 on AVX2 (a long vmaskmov chain) or VF64 on
AVX-512 (a long kshiftr chain).
Charge an additive (DataParts - MaskParts) * 4 for the extra parts:
* getMaskedMemoryOpCost and getGatherScatterOpCost on all subtargets,
gated on a variable (runtime) mask -- a constant/all-ones mask is
folded by codegen and needs no per-part sub-mask, so it pays none
(matching how the gather/scatter cost already threads VariableMask).
The charge only fires on genuinely predicated memory ops, so it is
safe to apply everywhere and it fixes the AVX2 over-widening reported
on the MaximizeBandwidth enablement, not just the AVX-512 case.
* getCmpSelInstrCost gated to AVX-512 only. A vector select is a
pervasive primitive; on SSE/AVX its blend is already priced by the
per-part tables and adding the fanout over-penalizes ordinary,
non-predicated-memory loops and needlessly shrinks their VF (observed
on baseline SSE2). On AVX-512 the k-register -> wide-data kshiftr is a
real cost the tables miss, so the charge is kept there.
The penalty is zero whenever data and mask occupy the same number of
register parts, so maskless and single-part reductions -- the
MaximizeBandwidth throughput wins -- are unaffected. With this the
motivating conditional-store loop selects VF16 on AVX-512 (no kshiftr)
and VF4 on AVX2 (no over-wide vmaskmov chain), so no generic
vectorizer-level cost floor is required.
---
%t | 25 +
.../lib/Target/X86/X86TargetTransformInfo.cpp | 100 ++-
llvm/test/Analysis/CostModel/X86/fptoi_sat.ll | 12 +-
.../X86/gather-scatter-mask-expansion.ll | 40 ++
.../X86/masked-intrinsic-cost-inseltpoison.ll | 108 +--
.../CostModel/X86/masked-intrinsic-cost.ll | 108 +--
.../X86/masked-mem-mask-expansion.ll | 91 +++
.../CostModel/X86/select-mask-expansion.ll | 63 ++
.../masked-gather-i32-with-i8-index.ll | 8 +-
.../masked-gather-i64-with-i8-index.ll | 12 +-
.../X86/CostModel/masked-load-i16.ll | 2 +-
.../X86/CostModel/masked-load-i32.ll | 12 +-
.../X86/CostModel/masked-load-i64.ll | 18 +-
.../masked-scatter-i32-with-i8-index.ll | 4 +-
.../masked-scatter-i64-with-i8-index.ll | 6 +-
.../X86/CostModel/masked-store-i16.ll | 2 +-
.../X86/CostModel/masked-store-i32.ll | 12 +-
.../X86/CostModel/masked-store-i64.ll | 18 +-
.../LoopVectorize/X86/masked_load_store.ll | 635 ++++++++++++------
.../X86/gep-nodes-with-non-gep-inst.ll | 26 +-
.../SLPVectorizer/X86/pr47629-inseltpoison.ll | 58 +-
.../Transforms/SLPVectorizer/X86/pr47629.ll | 58 +-
...masked-loads-consecutive-loads-same-ptr.ll | 11 +-
.../X86/scatter-vectorize-reused-pointer.ll | 12 +-
24 files changed, 1018 insertions(+), 423 deletions(-)
create mode 100644 %t
create mode 100644 llvm/test/Analysis/CostModel/X86/gather-scatter-mask-expansion.ll
create mode 100644 llvm/test/Analysis/CostModel/X86/masked-mem-mask-expansion.ll
create mode 100644 llvm/test/Analysis/CostModel/X86/select-mask-expansion.ll
diff --git a/%t b/%t
new file mode 100644
index 0000000000000..1d4e829e532b9
--- /dev/null
+++ b/%t
@@ -0,0 +1,25 @@
+--- !Passed
+Pass: slp-vectorizer
+Name: StoresVectorized
+Function: test
+Args:
+ - String: 'Stores SLP vectorized with cost '
+ - Cost: '-7'
+ - String: ' and with tree size '
+ - TreeSize: '5'
+...
+--- !Missed
+Pass: slp-vectorizer
+Name: NotPossible
+Function: test
+Args:
+ - String: 'Cannot SLP vectorize list: only 2 elements of buildvector, trying reduction first.'
+...
+--- !Missed
+Pass: slp-vectorizer
+Name: NotPossible
+Function: test
+Args:
+ - String: 'Cannot SLP vectorize list: vectorization was impossible'
+ - String: ' with available vectorization factors'
+...
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
index 8e0cf1fc5a153..3b1f0abdd4e86 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
@@ -3492,6 +3492,40 @@ InstructionCost X86TTIImpl::getCastInstrCost(unsigned Opcode, Type *Dst,
BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I));
}
+// Additive cost of "predicate fanout" (mask expansion) for a predicated op
+// whose value/data operand legalizes into more register parts than its mask.
+//
+// A predicated op (select, masked load/store, gather/scatter) produces its
+// mask as a compact <N x i1>: a compare result kept in one k-register on
+// AVX-512, or a single narrow vector on SSE/AVX. When the data type legalizes
+// into more register parts than that mask, codegen has to materialize a
+// sub-mask for every extra data part -- a kshiftr out of the k-register on
+// AVX-512, or a vector unpack/sign-extend on SSE/AVX -- each feeding a separate
+// per-part op. The per-part cost tables price the op linearly in data parts
+// and miss this fanout, so per-lane cost stays flat as the type widens; with
+// MaximizeBandwidth sizing VF from the narrowest type, nothing then penalizes
+// over-widening and predicated loops blow up.
+//
+// Charge FanoutCostPerPart per extra part. Two caveats worth stating:
+// - The metric assumes the i1 mask legalizes into no more parts than the
+// data; that holds on X86 (masks come from compares, kept compact). A
+// MaskParts of 0 (unexpected legalization) disables the charge.
+// - The constant is not an absolute latency. It only has to be large enough
+// that the part-matched VF beats the loop vectorizer's tie-break toward the
+// widest VF (a value of 2 tied and lost). It is orthogonal to the
+// promotion / expansion mask shuffles priced in getMaskedMemoryOpCost,
+// which cover data promotion and padding the mask to the legalized element
+// count -- not per-part sub-mask distribution.
+static InstructionCost getPredicateFanoutCost(const X86TTIImpl &TTI,
+ Type *DataVTy, Type *MaskVTy) {
+ constexpr unsigned FanoutCostPerPart = 4;
+ unsigned DataParts = TTI.getNumberOfParts(DataVTy);
+ unsigned MaskParts = TTI.getNumberOfParts(MaskVTy);
+ if (DataParts > MaskParts && MaskParts > 0)
+ return InstructionCost((DataParts - MaskParts) * FanoutCostPerPart);
+ return InstructionCost(0);
+}
+
InstructionCost X86TTIImpl::getCmpSelInstrCost(
unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info,
@@ -3509,6 +3543,18 @@ InstructionCost X86TTIImpl::getCmpSelInstrCost(
int ISD = TLI->InstructionOpcodeToISD(Opcode);
assert(ISD && "Invalid opcode");
+ // Mask-expansion (predicate fanout); see getPredicateFanoutCost. Gated to
+ // AVX-512 on purpose: a vector select is pervasive and on SSE/AVX its blend
+ // mask is already priced by the per-part tables, so charging the fanout there
+ // over-penalizes ordinary (non predicated-memory) loops and shrinks their VF.
+ // On AVX-512 the k-register -> wide-data kshiftr is a real cost the tables
+ // miss. The masked memory / gather / scatter fanout is charged on all
+ // subtargets because it fires only on genuinely predicated memory ops.
+ InstructionCost MaskExpCost = 0;
+ if (Opcode == Instruction::Select && ST->hasAVX512() &&
+ isa<VectorType>(ValTy) && isa_and_nonnull<VectorType>(CondTy))
+ MaskExpCost = getPredicateFanoutCost(*this, ValTy, CondTy);
+
InstructionCost ExtraCost = 0;
if (Opcode == Instruction::ICmp || Opcode == Instruction::FCmp) {
// Some vector comparison predicates cost extra instructions.
@@ -3735,52 +3781,52 @@ InstructionCost X86TTIImpl::getCmpSelInstrCost(
if (ST->useSLMArithCosts())
if (const auto *Entry = CostTableLookup(SLMCostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasBWI())
if (const auto *Entry = CostTableLookup(AVX512BWCostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasAVX512())
if (const auto *Entry = CostTableLookup(AVX512CostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasAVX2())
if (const auto *Entry = CostTableLookup(AVX2CostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasXOP())
if (const auto *Entry = CostTableLookup(XOPCostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasAVX())
if (const auto *Entry = CostTableLookup(AVX1CostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasSSE42())
if (const auto *Entry = CostTableLookup(SSE42CostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasSSE41())
if (const auto *Entry = CostTableLookup(SSE41CostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasSSE2())
if (const auto *Entry = CostTableLookup(SSE2CostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasSSE1())
if (const auto *Entry = CostTableLookup(SSE1CostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
// Assume a 3cy latency for fp select ops.
if (CostKind == TTI::TCK_Latency && Opcode == Instruction::Select)
@@ -3788,7 +3834,8 @@ InstructionCost X86TTIImpl::getCmpSelInstrCost(
return 3;
return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
- Op1Info, Op2Info, I);
+ Op1Info, Op2Info, I) +
+ MaskExpCost;
}
unsigned X86TTIImpl::getAtomicMemIntrinsicMaxElementSize() const { return 16; }
@@ -5679,12 +5726,22 @@ X86TTIImpl::getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA,
CostKind, {}, 0, MaskTy);
}
+ // Mask-expansion (predicate fanout); see getPredicateFanoutCost. Only a
+ // variable predicate needs a per-part sub-mask materialized; a constant /
+ // all-ones mask is folded by CodeGen and pays no fanout.
+ InstructionCost MaskExpCost = 0;
+ if (MICA.getVariableMask()) {
+ auto *MaskVecTy =
+ FixedVectorType::get(Type::getInt1Ty(SrcVTy->getContext()), NumElem);
+ MaskExpCost = getPredicateFanoutCost(*this, SrcVTy, MaskVecTy);
+ }
+
// Pre-AVX512 - each maskmov load costs 2 + store costs ~8.
if (!ST->hasAVX512())
- return Cost + LT.first * (IsLoad ? 2 : 8);
+ return Cost + LT.first * (IsLoad ? 2 : 8) + MaskExpCost;
- // AVX-512 masked load/store is cheaper
- return Cost + LT.first;
+ // AVX-512 masked load/store is cheaper.
+ return Cost + LT.first + MaskExpCost;
}
InstructionCost X86TTIImpl::getPointersChainCost(
@@ -6701,8 +6758,19 @@ X86TTIImpl::getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA,
assert(SrcVTy->isVectorTy() && "Unexpected data type for Gather/Scatter");
unsigned AddressSpace = MICA.getAddressSpace();
- return getGSVectorCost(Opcode, CostKind, SrcVTy, Ptr, Alignment,
- AddressSpace);
+ InstructionCost Cost =
+ getGSVectorCost(Opcode, CostKind, SrcVTy, Ptr, Alignment, AddressSpace);
+
+ // Mask-expansion (predicate fanout); see getPredicateFanoutCost. Only a
+ // variable predicate needs a per-part sub-mask materialized; a constant /
+ // all-ones mask is folded by CodeGen and pays no fanout.
+ if (MICA.getVariableMask()) {
+ auto *MaskVecTy =
+ FixedVectorType::get(Type::getInt1Ty(SrcVTy->getContext()),
+ cast<FixedVectorType>(SrcVTy)->getNumElements());
+ Cost += getPredicateFanoutCost(*this, SrcVTy, MaskVecTy);
+ }
+ return Cost;
}
bool X86TTIImpl::isLSRCostLess(const TargetTransformInfo::LSRCost &C1,
diff --git a/llvm/test/Analysis/CostModel/X86/fptoi_sat.ll b/llvm/test/Analysis/CostModel/X86/fptoi_sat.ll
index 41bf88b1ec316..e5df2badfeb34 100644
--- a/llvm/test/Analysis/CostModel/X86/fptoi_sat.ll
+++ b/llvm/test/Analysis/CostModel/X86/fptoi_sat.ll
@@ -512,7 +512,7 @@ define void @casts() {
; AVX512F-NEXT: Cost Model: Found an estimated cost of 11 for instruction: %v16f32u16 = call <16 x i16> @llvm.fptoui.sat.v16i16.v16f32(<16 x float> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 11 for instruction: %v16f32s32 = call <16 x i32> @llvm.fptosi.sat.v16i32.v16f32(<16 x float> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %v16f32u32 = call <16 x i32> @llvm.fptoui.sat.v16i32.v16f32(<16 x float> undef)
-; AVX512F-NEXT: Cost Model: Found an estimated cost of 72 for instruction: %v16f32s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f32(<16 x float> undef)
+; AVX512F-NEXT: Cost Model: Found an estimated cost of 76 for instruction: %v16f32s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f32(<16 x float> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 69 for instruction: %v16f32u64 = call <16 x i64> @llvm.fptoui.sat.v16i64.v16f32(<16 x float> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 49 for instruction: %v16f64s1 = call <16 x i1> @llvm.fptosi.sat.v16i1.v16f64(<16 x double> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 15 for instruction: %v16f64u1 = call <16 x i1> @llvm.fptoui.sat.v16i1.v16f64(<16 x double> undef)
@@ -522,7 +522,7 @@ define void @casts() {
; AVX512F-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %v16f64u16 = call <16 x i16> @llvm.fptoui.sat.v16i16.v16f64(<16 x double> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 18 for instruction: %v16f64s32 = call <16 x i32> @llvm.fptosi.sat.v16i32.v16f64(<16 x double> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 15 for instruction: %v16f64u32 = call <16 x i32> @llvm.fptoui.sat.v16i32.v16f64(<16 x double> undef)
-; AVX512F-NEXT: Cost Model: Found an estimated cost of 76 for instruction: %v16f64s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f64(<16 x double> undef)
+; AVX512F-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v16f64s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f64(<16 x double> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 72 for instruction: %v16f64u64 = call <16 x i64> @llvm.fptoui.sat.v16i64.v16f64(<16 x double> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
;
@@ -615,7 +615,7 @@ define void @casts() {
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 11 for instruction: %v16f32u16 = call <16 x i16> @llvm.fptoui.sat.v16i16.v16f32(<16 x float> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 11 for instruction: %v16f32s32 = call <16 x i32> @llvm.fptosi.sat.v16i32.v16f32(<16 x float> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %v16f32u32 = call <16 x i32> @llvm.fptoui.sat.v16i32.v16f32(<16 x float> undef)
-; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 14 for instruction: %v16f32s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f32(<16 x float> undef)
+; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 18 for instruction: %v16f32s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f32(<16 x float> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 11 for instruction: %v16f32u64 = call <16 x i64> @llvm.fptoui.sat.v16i64.v16f32(<16 x float> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 49 for instruction: %v16f64s1 = call <16 x i1> @llvm.fptosi.sat.v16i1.v16f64(<16 x double> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 15 for instruction: %v16f64u1 = call <16 x i1> @llvm.fptoui.sat.v16i1.v16f64(<16 x double> undef)
@@ -625,7 +625,7 @@ define void @casts() {
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %v16f64u16 = call <16 x i16> @llvm.fptoui.sat.v16i16.v16f64(<16 x double> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 18 for instruction: %v16f64s32 = call <16 x i32> @llvm.fptosi.sat.v16i32.v16f64(<16 x double> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 15 for instruction: %v16f64u32 = call <16 x i32> @llvm.fptoui.sat.v16i32.v16f64(<16 x double> undef)
-; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 18 for instruction: %v16f64s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f64(<16 x double> undef)
+; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 22 for instruction: %v16f64s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f64(<16 x double> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 14 for instruction: %v16f64u64 = call <16 x i64> @llvm.fptoui.sat.v16i64.v16f64(<16 x double> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
;
@@ -1054,7 +1054,7 @@ define void @fp16() {
; AVX512F-NEXT: Cost Model: Found an estimated cost of 178 for instruction: %v16f16u16 = call <16 x i16> @llvm.fptoui.sat.v16i16.v16f16(<16 x half> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 178 for instruction: %v16f16s32 = call <16 x i32> @llvm.fptosi.sat.v16i32.v16f16(<16 x half> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 176 for instruction: %v16f16u32 = call <16 x i32> @llvm.fptoui.sat.v16i32.v16f16(<16 x half> undef)
-; AVX512F-NEXT: Cost Model: Found an estimated cost of 186 for instruction: %v16f16s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f16(<16 x half> undef)
+; AVX512F-NEXT: Cost Model: Found an estimated cost of 190 for instruction: %v16f16s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f16(<16 x half> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 183 for instruction: %v16f16u64 = call <16 x i64> @llvm.fptoui.sat.v16i64.v16f16(<16 x half> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
;
@@ -1107,7 +1107,7 @@ define void @fp16() {
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 178 for instruction: %v16f16u16 = call <16 x i16> @llvm.fptoui.sat.v16i16.v16f16(<16 x half> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 178 for instruction: %v16f16s32 = call <16 x i32> @llvm.fptosi.sat.v16i32.v16f16(<16 x half> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 176 for instruction: %v16f16u32 = call <16 x i32> @llvm.fptoui.sat.v16i32.v16f16(<16 x half> undef)
-; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 186 for instruction: %v16f16s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f16(<16 x half> undef)
+; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 190 for instruction: %v16f16s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f16(<16 x half> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 183 for instruction: %v16f16u64 = call <16 x i64> @llvm.fptoui.sat.v16i64.v16f16(<16 x half> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
;
diff --git a/llvm/test/Analysis/CostModel/X86/gather-scatter-mask-expansion.ll b/llvm/test/Analysis/CostModel/X86/gather-scatter-mask-expansion.ll
new file mode 100644
index 0000000000000..1e7553185c2fa
--- /dev/null
+++ b/llvm/test/Analysis/CostModel/X86/gather-scatter-mask-expansion.ll
@@ -0,0 +1,40 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+avx2 | FileCheck %s -check-prefixes=CHECK,AVX2
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+avx512f | FileCheck %s -check-prefixes=CHECK,AVX512
+
+; A masked gather/scatter whose data type legalizes into more register parts
+; than its <N x i1> mask needs extra kshiftr instructions on AVX-512 to extract
+; a sub-mask for each data part. That fanout is charged only on AVX-512, where
+; the mask lives in a single k-register; AVX2 keeps a full-width vector mask and
+; is not charged the extra term (its per-part gather cost already dominates).
+
+define <32 x i32> @gather_v32i32(<32 x ptr> %ptrs, <32 x i1> %m, <32 x i32> %pass) {
+; AVX2-LABEL: 'gather_v32i32'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 109 for instruction: %g = call <32 x i32> @llvm.masked.gather.v32i32.v32p0(<32 x ptr> align 4 %ptrs, <32 x i1> %m, <32 x i32> %pass)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <32 x i32> %g
+;
+; AVX512-LABEL: 'gather_v32i32'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %g = call <32 x i32> @llvm.masked.gather.v32i32.v32p0(<32 x ptr> align 4 %ptrs, <32 x i1> %m, <32 x i32> %pass)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <32 x i32> %g
+;
+ %g = call <32 x i32> @llvm.masked.gather.v32i32.v32p0(<32 x ptr> %ptrs, i32 4, <32 x i1> %m, <32 x i32> %pass)
+ ret <32 x i32> %g
+}
+
+define void @scatter_v32i32(<32 x i32> %v, <32 x ptr> %ptrs, <32 x i1> %m) {
+; AVX2-LABEL: 'scatter_v32i32'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 109 for instruction: call void @llvm.masked.scatter.v32i32.v32p0(<32 x i32> %v, <32 x ptr> align 4 %ptrs, <32 x i1> %m)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; AVX512-LABEL: 'scatter_v32i32'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 40 for instruction: call void @llvm.masked.scatter.v32i32.v32p0(<32 x i32> %v, <32 x ptr> align 4 %ptrs, <32 x i1> %m)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+ call void @llvm.masked.scatter.v32i32.v32p0(<32 x i32> %v, <32 x ptr> %ptrs, i32 4, <32 x i1> %m)
+ ret void
+}
+
+declare <32 x i32> @llvm.masked.gather.v32i32.v32p0(<32 x ptr>, i32, <32 x i1>, <32 x i32>)
+declare void @llvm.masked.scatter.v32i32.v32p0(<32 x i32>, <32 x ptr>, i32, <32 x i1>)
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK: {{.*}}
diff --git a/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost-inseltpoison.ll b/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost-inseltpoison.ll
index f835a8aedbbcc..8c26af02c1256 100644
--- a/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost-inseltpoison.ll
+++ b/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost-inseltpoison.ll
@@ -128,22 +128,22 @@ define i32 @masked_load(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; SSE42-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 0
;
; AVX-LABEL: 'masked_load'
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V8F64 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V7F64 = call <7 x double> @llvm.masked.load.v7f64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V6F64 = call <6 x double> @llvm.masked.load.v6f64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V5F64 = call <5 x double> @llvm.masked.load.v5f64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V8F64 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V7F64 = call <7 x double> @llvm.masked.load.v7f64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V6F64 = call <6 x double> @llvm.masked.load.v6f64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V5F64 = call <5 x double> @llvm.masked.load.v5f64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x double> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V4F64 = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 1 undef, <4 x i1> %m4, <4 x double> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V3F64 = call <3 x double> @llvm.masked.load.v3f64.p0(ptr align 1 undef, <3 x i1> %m3, <3 x double> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V2F64 = call <2 x double> @llvm.masked.load.v2f64.p0(ptr align 1 undef, <2 x i1> %m2, <2 x double> undef)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1F64 = call <1 x double> @llvm.masked.load.v1f64.p0(ptr align 1 undef, <1 x i1> %m1, <1 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V16F32 = call <16 x float> @llvm.masked.load.v16f32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V15F32 = call <15 x float> @llvm.masked.load.v15f32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V14F32 = call <14 x float> @llvm.masked.load.v14f32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V13F32 = call <13 x float> @llvm.masked.load.v13f32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V12F32 = call <12 x float> @llvm.masked.load.v12f32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V11F32 = call <11 x float> @llvm.masked.load.v11f32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V10F32 = call <10 x float> @llvm.masked.load.v10f32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V9F32 = call <9 x float> @llvm.masked.load.v9f32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V16F32 = call <16 x float> @llvm.masked.load.v16f32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V15F32 = call <15 x float> @llvm.masked.load.v15f32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V14F32 = call <14 x float> @llvm.masked.load.v14f32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V13F32 = call <13 x float> @llvm.masked.load.v13f32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V12F32 = call <12 x float> @llvm.masked.load.v12f32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V11F32 = call <11 x float> @llvm.masked.load.v11f32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V10F32 = call <10 x float> @llvm.masked.load.v10f32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V9F32 = call <9 x float> @llvm.masked.load.v9f32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V8F32 = call <8 x float> @llvm.masked.load.v8f32.p0(ptr align 1 undef, <8 x i1> %m8, <8 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V7F32 = call <7 x float> @llvm.masked.load.v7f32.p0(ptr align 1 undef, <7 x i1> %m7, <7 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V6F32 = call <6 x float> @llvm.masked.load.v6f32.p0(ptr align 1 undef, <6 x i1> %m6, <6 x float> undef)
@@ -152,22 +152,22 @@ define i32 @masked_load(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; AVX-NEXT: Cost Model: Found costs of 3 for: %V3F32 = call <3 x float> @llvm.masked.load.v3f32.p0(ptr align 1 undef, <3 x i1> %m3, <3 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V2F32 = call <2 x float> @llvm.masked.load.v2f32.p0(ptr align 1 undef, <2 x i1> %m2, <2 x float> undef)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1F32 = call <1 x float> @llvm.masked.load.v1f32.p0(ptr align 1 undef, <1 x i1> %m1, <1 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V8I64 = call <8 x i64> @llvm.masked.load.v8i64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V7I64 = call <7 x i64> @llvm.masked.load.v7i64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V6I64 = call <6 x i64> @llvm.masked.load.v6i64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V5I64 = call <5 x i64> @llvm.masked.load.v5i64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V8I64 = call <8 x i64> @llvm.masked.load.v8i64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V7I64 = call <7 x i64> @llvm.masked.load.v7i64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V6I64 = call <6 x i64> @llvm.masked.load.v6i64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V5I64 = call <5 x i64> @llvm.masked.load.v5i64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V4I64 = call <4 x i64> @llvm.masked.load.v4i64.p0(ptr align 1 undef, <4 x i1> %m4, <4 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V3I64 = call <3 x i64> @llvm.masked.load.v3i64.p0(ptr align 1 undef, <3 x i1> %m3, <3 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V2I64 = call <2 x i64> @llvm.masked.load.v2i64.p0(ptr align 1 undef, <2 x i1> %m2, <2 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1I64 = call <1 x i64> @llvm.masked.load.v1i64.p0(ptr align 1 undef, <1 x i1> %m1, <1 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V16I32 = call <16 x i32> @llvm.masked.load.v16i32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V15I32 = call <15 x i32> @llvm.masked.load.v15i32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V14I32 = call <14 x i32> @llvm.masked.load.v14i32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V13I32 = call <13 x i32> @llvm.masked.load.v13i32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V12I32 = call <12 x i32> @llvm.masked.load.v12i32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V11I32 = call <11 x i32> @llvm.masked.load.v11i32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V10I32 = call <10 x i32> @llvm.masked.load.v10i32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V9I32 = call <9 x i32> @llvm.masked.load.v9i32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V16I32 = call <16 x i32> @llvm.masked.load.v16i32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V15I32 = call <15 x i32> @llvm.masked.load.v15i32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V14I32 = call <14 x i32> @llvm.masked.load.v14i32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V13I32 = call <13 x i32> @llvm.masked.load.v13i32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V12I32 = call <12 x i32> @llvm.masked.load.v12i32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V11I32 = call <11 x i32> @llvm.masked.load.v11i32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V10I32 = call <10 x i32> @llvm.masked.load.v10i32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V9I32 = call <9 x i32> @llvm.masked.load.v9i32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x i32> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V8I32 = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 undef, <8 x i1> %m8, <8 x i32> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V7I32 = call <7 x i32> @llvm.masked.load.v7i32.p0(ptr align 1 undef, <7 x i1> %m7, <7 x i32> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V6I32 = call <6 x i32> @llvm.masked.load.v6i32.p0(ptr align 1 undef, <6 x i1> %m6, <6 x i32> undef)
@@ -489,22 +489,22 @@ define i32 @masked_store(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; SSE42-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 0
;
; AVX-LABEL: 'masked_store'
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v8f64.p0(<8 x double> undef, ptr align 1 undef, <8 x i1> %m8)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v7f64.p0(<7 x double> undef, ptr align 1 undef, <7 x i1> %m7)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v6f64.p0(<6 x double> undef, ptr align 1 undef, <6 x i1> %m6)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v5f64.p0(<5 x double> undef, ptr align 1 undef, <5 x i1> %m5)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v8f64.p0(<8 x double> undef, ptr align 1 undef, <8 x i1> %m8)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v7f64.p0(<7 x double> undef, ptr align 1 undef, <7 x i1> %m7)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v6f64.p0(<6 x double> undef, ptr align 1 undef, <6 x i1> %m6)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v5f64.p0(<5 x double> undef, ptr align 1 undef, <5 x i1> %m5)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v4f64.p0(<4 x double> undef, ptr align 1 undef, <4 x i1> %m4)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v3f64.p0(<3 x double> undef, ptr align 1 undef, <3 x i1> %m3)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v2f64.p0(<2 x double> undef, ptr align 1 undef, <2 x i1> %m2)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:3 SizeLat:3 for: call void @llvm.masked.store.v1f64.p0(<1 x double> undef, ptr align 1 undef, <1 x i1> %m1)
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v16f32.p0(<16 x float> undef, ptr align 1 undef, <16 x i1> %m16)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v15f32.p0(<15 x float> undef, ptr align 1 undef, <15 x i1> %m15)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v14f32.p0(<14 x float> undef, ptr align 1 undef, <14 x i1> %m14)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v13f32.p0(<13 x float> undef, ptr align 1 undef, <13 x i1> %m13)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v12f32.p0(<12 x float> undef, ptr align 1 undef, <12 x i1> %m12)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v11f32.p0(<11 x float> undef, ptr align 1 undef, <11 x i1> %m11)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v10f32.p0(<10 x float> undef, ptr align 1 undef, <10 x i1> %m10)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v9f32.p0(<9 x float> undef, ptr align 1 undef, <9 x i1> %m9)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v16f32.p0(<16 x float> undef, ptr align 1 undef, <16 x i1> %m16)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v15f32.p0(<15 x float> undef, ptr align 1 undef, <15 x i1> %m15)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v14f32.p0(<14 x float> undef, ptr align 1 undef, <14 x i1> %m14)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v13f32.p0(<13 x float> undef, ptr align 1 undef, <13 x i1> %m13)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v12f32.p0(<12 x float> undef, ptr align 1 undef, <12 x i1> %m12)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v11f32.p0(<11 x float> undef, ptr align 1 undef, <11 x i1> %m11)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v10f32.p0(<10 x float> undef, ptr align 1 undef, <10 x i1> %m10)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v9f32.p0(<9 x float> undef, ptr align 1 undef, <9 x i1> %m9)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v8f32.p0(<8 x float> undef, ptr align 1 undef, <8 x i1> %m8)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v7f32.p0(<7 x float> undef, ptr align 1 undef, <7 x i1> %m7)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v6f32.p0(<6 x float> undef, ptr align 1 undef, <6 x i1> %m6)
@@ -513,22 +513,22 @@ define i32 @masked_store(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v3f32.p0(<3 x float> undef, ptr align 1 undef, <3 x i1> %m3)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v2f32.p0(<2 x float> undef, ptr align 1 undef, <2 x i1> %m2)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:3 SizeLat:3 for: call void @llvm.masked.store.v1f32.p0(<1 x float> undef, ptr align 1 undef, <1 x i1> %m1)
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v8i64.p0(<8 x i64> undef, ptr align 1 undef, <8 x i1> %m8)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v7i64.p0(<7 x i64> undef, ptr align 1 undef, <7 x i1> %m7)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v6i64.p0(<6 x i64> undef, ptr align 1 undef, <6 x i1> %m6)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v5i64.p0(<5 x i64> undef, ptr align 1 undef, <5 x i1> %m5)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v8i64.p0(<8 x i64> undef, ptr align 1 undef, <8 x i1> %m8)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v7i64.p0(<7 x i64> undef, ptr align 1 undef, <7 x i1> %m7)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v6i64.p0(<6 x i64> undef, ptr align 1 undef, <6 x i1> %m6)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v5i64.p0(<5 x i64> undef, ptr align 1 undef, <5 x i1> %m5)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v4i64.p0(<4 x i64> undef, ptr align 1 undef, <4 x i1> %m4)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v3i64.p0(<3 x i64> undef, ptr align 1 undef, <3 x i1> %m3)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v2i64.p0(<2 x i64> undef, ptr align 1 undef, <2 x i1> %m2)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:3 SizeLat:3 for: call void @llvm.masked.store.v1i64.p0(<1 x i64> undef, ptr align 1 undef, <1 x i1> %m1)
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v16i32.p0(<16 x i32> undef, ptr align 1 undef, <16 x i1> %m16)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v15i32.p0(<15 x i32> undef, ptr align 1 undef, <15 x i1> %m15)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v14i32.p0(<14 x i32> undef, ptr align 1 undef, <14 x i1> %m14)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v13i32.p0(<13 x i32> undef, ptr align 1 undef, <13 x i1> %m13)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v12i32.p0(<12 x i32> undef, ptr align 1 undef, <12 x i1> %m12)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v11i32.p0(<11 x i32> undef, ptr align 1 undef, <11 x i1> %m11)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v10i32.p0(<10 x i32> undef, ptr align 1 undef, <10 x i1> %m10)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v9i32.p0(<9 x i32> undef, ptr align 1 undef, <9 x i1> %m9)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v16i32.p0(<16 x i32> undef, ptr align 1 undef, <16 x i1> %m16)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v15i32.p0(<15 x i32> undef, ptr align 1 undef, <15 x i1> %m15)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v14i32.p0(<14 x i32> undef, ptr align 1 undef, <14 x i1> %m14)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v13i32.p0(<13 x i32> undef, ptr align 1 undef, <13 x i1> %m13)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v12i32.p0(<12 x i32> undef, ptr align 1 undef, <12 x i1> %m12)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v11i32.p0(<11 x i32> undef, ptr align 1 undef, <11 x i1> %m11)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v10i32.p0(<10 x i32> undef, ptr align 1 undef, <10 x i1> %m10)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v9i32.p0(<9 x i32> undef, ptr align 1 undef, <9 x i1> %m9)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v8i32.p0(<8 x i32> undef, ptr align 1 undef, <8 x i1> %m8)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v7i32.p0(<7 x i32> undef, ptr align 1 undef, <7 x i1> %m7)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v6i32.p0(<6 x i32> undef, ptr align 1 undef, <6 x i1> %m6)
@@ -840,19 +840,19 @@ define i32 @masked_gather(<1 x i1> %m1, <2 x i1> %m2, <4 x i1> %m4, <8 x i1> %m8
; AVX2-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 0
;
; SKL-LABEL: 'masked_gather'
-; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8F64 = call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x double> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:16 CodeSize:6 Lat:40 SizeLat:16 for: %V8F64 = call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x double> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4F64 = call <4 x double> @llvm.masked.gather.v4f64.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x double> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2F64 = call <2 x double> @llvm.masked.gather.v2f64.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x double> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1F64 = call <1 x double> @llvm.masked.gather.v1f64.v1p0(<1 x ptr> align 1 undef, <1 x i1> %m1, <1 x double> undef)
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %V16F32 = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %V16F32 = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8F32 = call <8 x float> @llvm.masked.gather.v8f32.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4F32 = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2F32 = call <2 x float> @llvm.masked.gather.v2f32.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x float> undef)
-; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8I64 = call <8 x i64> @llvm.masked.gather.v8i64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x i64> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:16 CodeSize:6 Lat:40 SizeLat:16 for: %V8I64 = call <8 x i64> @llvm.masked.gather.v8i64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x i64> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4I64 = call <4 x i64> @llvm.masked.gather.v4i64.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x i64> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2I64 = call <2 x i64> @llvm.masked.gather.v2i64.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x i64> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1I64 = call <1 x i64> @llvm.masked.gather.v1i64.v1p0(<1 x ptr> align 1 undef, <1 x i1> %m1, <1 x i64> undef)
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %V16I32 = call <16 x i32> @llvm.masked.gather.v16i32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x i32> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %V16I32 = call <16 x i32> @llvm.masked.gather.v16i32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x i32> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8I32 = call <8 x i32> @llvm.masked.gather.v8i32.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x i32> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4I32 = call <4 x i32> @llvm.masked.gather.v4i32.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x i32> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2I32 = call <2 x i32> @llvm.masked.gather.v2i32.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x i32> undef)
@@ -1953,7 +1953,7 @@ define <16 x float> @test_gather_16f32_var_mask(ptr %base, <16 x i32> %ind, <16
; SKL-LABEL: 'test_gather_16f32_var_mask'
; SKL-NEXT: Cost Model: Found costs of RThru:10 CodeSize:1 Lat:1 SizeLat:1 for: %sext_ind = sext <16 x i32> %ind to <16 x i64>
; SKL-NEXT: Cost Model: Found costs of 0 for: %gep.v = getelementptr float, ptr %base, <16 x i64> %sext_ind
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret <16 x float> %res
;
; AVX512-LABEL: 'test_gather_16f32_var_mask'
@@ -1997,7 +1997,7 @@ define <16 x float> @test_gather_16f32_ra_var_mask(<16 x ptr> %ptrs, <16 x i32>
; SKL-LABEL: 'test_gather_16f32_ra_var_mask'
; SKL-NEXT: Cost Model: Found costs of RThru:10 CodeSize:1 Lat:1 SizeLat:1 for: %sext_ind = sext <16 x i32> %ind to <16 x i64>
; SKL-NEXT: Cost Model: Found costs of 0 for: %gep.v = getelementptr float, <16 x ptr> %ptrs, <16 x i64> %sext_ind
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret <16 x float> %res
;
; AVX512-LABEL: 'test_gather_16f32_ra_var_mask'
diff --git a/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost.ll b/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost.ll
index ed1b534fac8f8..7b9f2aa80b808 100644
--- a/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost.ll
+++ b/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost.ll
@@ -128,22 +128,22 @@ define i32 @masked_load(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; SSE42-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 0
;
; AVX-LABEL: 'masked_load'
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V8F64 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V7F64 = call <7 x double> @llvm.masked.load.v7f64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V6F64 = call <6 x double> @llvm.masked.load.v6f64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V5F64 = call <5 x double> @llvm.masked.load.v5f64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V8F64 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V7F64 = call <7 x double> @llvm.masked.load.v7f64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V6F64 = call <6 x double> @llvm.masked.load.v6f64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V5F64 = call <5 x double> @llvm.masked.load.v5f64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x double> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V4F64 = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 1 undef, <4 x i1> %m4, <4 x double> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V3F64 = call <3 x double> @llvm.masked.load.v3f64.p0(ptr align 1 undef, <3 x i1> %m3, <3 x double> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V2F64 = call <2 x double> @llvm.masked.load.v2f64.p0(ptr align 1 undef, <2 x i1> %m2, <2 x double> undef)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1F64 = call <1 x double> @llvm.masked.load.v1f64.p0(ptr align 1 undef, <1 x i1> %m1, <1 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V16F32 = call <16 x float> @llvm.masked.load.v16f32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V15F32 = call <15 x float> @llvm.masked.load.v15f32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V14F32 = call <14 x float> @llvm.masked.load.v14f32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V13F32 = call <13 x float> @llvm.masked.load.v13f32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V12F32 = call <12 x float> @llvm.masked.load.v12f32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V11F32 = call <11 x float> @llvm.masked.load.v11f32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V10F32 = call <10 x float> @llvm.masked.load.v10f32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V9F32 = call <9 x float> @llvm.masked.load.v9f32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V16F32 = call <16 x float> @llvm.masked.load.v16f32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V15F32 = call <15 x float> @llvm.masked.load.v15f32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V14F32 = call <14 x float> @llvm.masked.load.v14f32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V13F32 = call <13 x float> @llvm.masked.load.v13f32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V12F32 = call <12 x float> @llvm.masked.load.v12f32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V11F32 = call <11 x float> @llvm.masked.load.v11f32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V10F32 = call <10 x float> @llvm.masked.load.v10f32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V9F32 = call <9 x float> @llvm.masked.load.v9f32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V8F32 = call <8 x float> @llvm.masked.load.v8f32.p0(ptr align 1 undef, <8 x i1> %m8, <8 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V7F32 = call <7 x float> @llvm.masked.load.v7f32.p0(ptr align 1 undef, <7 x i1> %m7, <7 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V6F32 = call <6 x float> @llvm.masked.load.v6f32.p0(ptr align 1 undef, <6 x i1> %m6, <6 x float> undef)
@@ -152,22 +152,22 @@ define i32 @masked_load(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; AVX-NEXT: Cost Model: Found costs of 3 for: %V3F32 = call <3 x float> @llvm.masked.load.v3f32.p0(ptr align 1 undef, <3 x i1> %m3, <3 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V2F32 = call <2 x float> @llvm.masked.load.v2f32.p0(ptr align 1 undef, <2 x i1> %m2, <2 x float> undef)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1F32 = call <1 x float> @llvm.masked.load.v1f32.p0(ptr align 1 undef, <1 x i1> %m1, <1 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V8I64 = call <8 x i64> @llvm.masked.load.v8i64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V7I64 = call <7 x i64> @llvm.masked.load.v7i64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V6I64 = call <6 x i64> @llvm.masked.load.v6i64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V5I64 = call <5 x i64> @llvm.masked.load.v5i64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V8I64 = call <8 x i64> @llvm.masked.load.v8i64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V7I64 = call <7 x i64> @llvm.masked.load.v7i64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V6I64 = call <6 x i64> @llvm.masked.load.v6i64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V5I64 = call <5 x i64> @llvm.masked.load.v5i64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V4I64 = call <4 x i64> @llvm.masked.load.v4i64.p0(ptr align 1 undef, <4 x i1> %m4, <4 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V3I64 = call <3 x i64> @llvm.masked.load.v3i64.p0(ptr align 1 undef, <3 x i1> %m3, <3 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V2I64 = call <2 x i64> @llvm.masked.load.v2i64.p0(ptr align 1 undef, <2 x i1> %m2, <2 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1I64 = call <1 x i64> @llvm.masked.load.v1i64.p0(ptr align 1 undef, <1 x i1> %m1, <1 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V16I32 = call <16 x i32> @llvm.masked.load.v16i32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V15I32 = call <15 x i32> @llvm.masked.load.v15i32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V14I32 = call <14 x i32> @llvm.masked.load.v14i32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V13I32 = call <13 x i32> @llvm.masked.load.v13i32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V12I32 = call <12 x i32> @llvm.masked.load.v12i32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V11I32 = call <11 x i32> @llvm.masked.load.v11i32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V10I32 = call <10 x i32> @llvm.masked.load.v10i32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V9I32 = call <9 x i32> @llvm.masked.load.v9i32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V16I32 = call <16 x i32> @llvm.masked.load.v16i32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V15I32 = call <15 x i32> @llvm.masked.load.v15i32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V14I32 = call <14 x i32> @llvm.masked.load.v14i32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V13I32 = call <13 x i32> @llvm.masked.load.v13i32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V12I32 = call <12 x i32> @llvm.masked.load.v12i32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V11I32 = call <11 x i32> @llvm.masked.load.v11i32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V10I32 = call <10 x i32> @llvm.masked.load.v10i32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V9I32 = call <9 x i32> @llvm.masked.load.v9i32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x i32> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V8I32 = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 undef, <8 x i1> %m8, <8 x i32> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V7I32 = call <7 x i32> @llvm.masked.load.v7i32.p0(ptr align 1 undef, <7 x i1> %m7, <7 x i32> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V6I32 = call <6 x i32> @llvm.masked.load.v6i32.p0(ptr align 1 undef, <6 x i1> %m6, <6 x i32> undef)
@@ -489,22 +489,22 @@ define i32 @masked_store(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; SSE42-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 0
;
; AVX-LABEL: 'masked_store'
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v8f64.p0(<8 x double> undef, ptr align 1 undef, <8 x i1> %m8)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v7f64.p0(<7 x double> undef, ptr align 1 undef, <7 x i1> %m7)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v6f64.p0(<6 x double> undef, ptr align 1 undef, <6 x i1> %m6)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v5f64.p0(<5 x double> undef, ptr align 1 undef, <5 x i1> %m5)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v8f64.p0(<8 x double> undef, ptr align 1 undef, <8 x i1> %m8)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v7f64.p0(<7 x double> undef, ptr align 1 undef, <7 x i1> %m7)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v6f64.p0(<6 x double> undef, ptr align 1 undef, <6 x i1> %m6)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v5f64.p0(<5 x double> undef, ptr align 1 undef, <5 x i1> %m5)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v4f64.p0(<4 x double> undef, ptr align 1 undef, <4 x i1> %m4)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v3f64.p0(<3 x double> undef, ptr align 1 undef, <3 x i1> %m3)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v2f64.p0(<2 x double> undef, ptr align 1 undef, <2 x i1> %m2)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:3 SizeLat:3 for: call void @llvm.masked.store.v1f64.p0(<1 x double> undef, ptr align 1 undef, <1 x i1> %m1)
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v16f32.p0(<16 x float> undef, ptr align 1 undef, <16 x i1> %m16)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v15f32.p0(<15 x float> undef, ptr align 1 undef, <15 x i1> %m15)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v14f32.p0(<14 x float> undef, ptr align 1 undef, <14 x i1> %m14)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v13f32.p0(<13 x float> undef, ptr align 1 undef, <13 x i1> %m13)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v12f32.p0(<12 x float> undef, ptr align 1 undef, <12 x i1> %m12)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v11f32.p0(<11 x float> undef, ptr align 1 undef, <11 x i1> %m11)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v10f32.p0(<10 x float> undef, ptr align 1 undef, <10 x i1> %m10)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v9f32.p0(<9 x float> undef, ptr align 1 undef, <9 x i1> %m9)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v16f32.p0(<16 x float> undef, ptr align 1 undef, <16 x i1> %m16)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v15f32.p0(<15 x float> undef, ptr align 1 undef, <15 x i1> %m15)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v14f32.p0(<14 x float> undef, ptr align 1 undef, <14 x i1> %m14)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v13f32.p0(<13 x float> undef, ptr align 1 undef, <13 x i1> %m13)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v12f32.p0(<12 x float> undef, ptr align 1 undef, <12 x i1> %m12)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v11f32.p0(<11 x float> undef, ptr align 1 undef, <11 x i1> %m11)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v10f32.p0(<10 x float> undef, ptr align 1 undef, <10 x i1> %m10)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v9f32.p0(<9 x float> undef, ptr align 1 undef, <9 x i1> %m9)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v8f32.p0(<8 x float> undef, ptr align 1 undef, <8 x i1> %m8)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v7f32.p0(<7 x float> undef, ptr align 1 undef, <7 x i1> %m7)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v6f32.p0(<6 x float> undef, ptr align 1 undef, <6 x i1> %m6)
@@ -513,22 +513,22 @@ define i32 @masked_store(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v3f32.p0(<3 x float> undef, ptr align 1 undef, <3 x i1> %m3)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v2f32.p0(<2 x float> undef, ptr align 1 undef, <2 x i1> %m2)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:3 SizeLat:3 for: call void @llvm.masked.store.v1f32.p0(<1 x float> undef, ptr align 1 undef, <1 x i1> %m1)
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v8i64.p0(<8 x i64> undef, ptr align 1 undef, <8 x i1> %m8)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v7i64.p0(<7 x i64> undef, ptr align 1 undef, <7 x i1> %m7)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v6i64.p0(<6 x i64> undef, ptr align 1 undef, <6 x i1> %m6)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v5i64.p0(<5 x i64> undef, ptr align 1 undef, <5 x i1> %m5)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v8i64.p0(<8 x i64> undef, ptr align 1 undef, <8 x i1> %m8)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v7i64.p0(<7 x i64> undef, ptr align 1 undef, <7 x i1> %m7)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v6i64.p0(<6 x i64> undef, ptr align 1 undef, <6 x i1> %m6)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v5i64.p0(<5 x i64> undef, ptr align 1 undef, <5 x i1> %m5)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v4i64.p0(<4 x i64> undef, ptr align 1 undef, <4 x i1> %m4)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v3i64.p0(<3 x i64> undef, ptr align 1 undef, <3 x i1> %m3)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v2i64.p0(<2 x i64> undef, ptr align 1 undef, <2 x i1> %m2)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:3 SizeLat:3 for: call void @llvm.masked.store.v1i64.p0(<1 x i64> undef, ptr align 1 undef, <1 x i1> %m1)
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v16i32.p0(<16 x i32> undef, ptr align 1 undef, <16 x i1> %m16)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v15i32.p0(<15 x i32> undef, ptr align 1 undef, <15 x i1> %m15)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v14i32.p0(<14 x i32> undef, ptr align 1 undef, <14 x i1> %m14)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v13i32.p0(<13 x i32> undef, ptr align 1 undef, <13 x i1> %m13)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v12i32.p0(<12 x i32> undef, ptr align 1 undef, <12 x i1> %m12)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v11i32.p0(<11 x i32> undef, ptr align 1 undef, <11 x i1> %m11)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v10i32.p0(<10 x i32> undef, ptr align 1 undef, <10 x i1> %m10)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v9i32.p0(<9 x i32> undef, ptr align 1 undef, <9 x i1> %m9)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v16i32.p0(<16 x i32> undef, ptr align 1 undef, <16 x i1> %m16)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v15i32.p0(<15 x i32> undef, ptr align 1 undef, <15 x i1> %m15)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v14i32.p0(<14 x i32> undef, ptr align 1 undef, <14 x i1> %m14)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v13i32.p0(<13 x i32> undef, ptr align 1 undef, <13 x i1> %m13)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v12i32.p0(<12 x i32> undef, ptr align 1 undef, <12 x i1> %m12)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v11i32.p0(<11 x i32> undef, ptr align 1 undef, <11 x i1> %m11)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v10i32.p0(<10 x i32> undef, ptr align 1 undef, <10 x i1> %m10)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v9i32.p0(<9 x i32> undef, ptr align 1 undef, <9 x i1> %m9)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v8i32.p0(<8 x i32> undef, ptr align 1 undef, <8 x i1> %m8)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v7i32.p0(<7 x i32> undef, ptr align 1 undef, <7 x i1> %m7)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v6i32.p0(<6 x i32> undef, ptr align 1 undef, <6 x i1> %m6)
@@ -840,19 +840,19 @@ define i32 @masked_gather(<1 x i1> %m1, <2 x i1> %m2, <4 x i1> %m4, <8 x i1> %m8
; AVX2-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 0
;
; SKL-LABEL: 'masked_gather'
-; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8F64 = call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x double> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:16 CodeSize:6 Lat:40 SizeLat:16 for: %V8F64 = call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x double> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4F64 = call <4 x double> @llvm.masked.gather.v4f64.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x double> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2F64 = call <2 x double> @llvm.masked.gather.v2f64.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x double> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1F64 = call <1 x double> @llvm.masked.gather.v1f64.v1p0(<1 x ptr> align 1 undef, <1 x i1> %m1, <1 x double> undef)
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %V16F32 = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %V16F32 = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8F32 = call <8 x float> @llvm.masked.gather.v8f32.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4F32 = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2F32 = call <2 x float> @llvm.masked.gather.v2f32.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x float> undef)
-; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8I64 = call <8 x i64> @llvm.masked.gather.v8i64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x i64> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:16 CodeSize:6 Lat:40 SizeLat:16 for: %V8I64 = call <8 x i64> @llvm.masked.gather.v8i64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x i64> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4I64 = call <4 x i64> @llvm.masked.gather.v4i64.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x i64> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2I64 = call <2 x i64> @llvm.masked.gather.v2i64.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x i64> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1I64 = call <1 x i64> @llvm.masked.gather.v1i64.v1p0(<1 x ptr> align 1 undef, <1 x i1> %m1, <1 x i64> undef)
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %V16I32 = call <16 x i32> @llvm.masked.gather.v16i32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x i32> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %V16I32 = call <16 x i32> @llvm.masked.gather.v16i32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x i32> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8I32 = call <8 x i32> @llvm.masked.gather.v8i32.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x i32> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4I32 = call <4 x i32> @llvm.masked.gather.v4i32.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x i32> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2I32 = call <2 x i32> @llvm.masked.gather.v2i32.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x i32> undef)
@@ -1953,7 +1953,7 @@ define <16 x float> @test_gather_16f32_var_mask(ptr %base, <16 x i32> %ind, <16
; SKL-LABEL: 'test_gather_16f32_var_mask'
; SKL-NEXT: Cost Model: Found costs of RThru:10 CodeSize:1 Lat:1 SizeLat:1 for: %sext_ind = sext <16 x i32> %ind to <16 x i64>
; SKL-NEXT: Cost Model: Found costs of 0 for: %gep.v = getelementptr float, ptr %base, <16 x i64> %sext_ind
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret <16 x float> %res
;
; AVX512-LABEL: 'test_gather_16f32_var_mask'
@@ -1997,7 +1997,7 @@ define <16 x float> @test_gather_16f32_ra_var_mask(<16 x ptr> %ptrs, <16 x i32>
; SKL-LABEL: 'test_gather_16f32_ra_var_mask'
; SKL-NEXT: Cost Model: Found costs of RThru:10 CodeSize:1 Lat:1 SizeLat:1 for: %sext_ind = sext <16 x i32> %ind to <16 x i64>
; SKL-NEXT: Cost Model: Found costs of 0 for: %gep.v = getelementptr float, <16 x ptr> %ptrs, <16 x i64> %sext_ind
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret <16 x float> %res
;
; AVX512-LABEL: 'test_gather_16f32_ra_var_mask'
diff --git a/llvm/test/Analysis/CostModel/X86/masked-mem-mask-expansion.ll b/llvm/test/Analysis/CostModel/X86/masked-mem-mask-expansion.ll
new file mode 100644
index 0000000000000..ab6623b6d2c90
--- /dev/null
+++ b/llvm/test/Analysis/CostModel/X86/masked-mem-mask-expansion.ll
@@ -0,0 +1,91 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+sse2 | FileCheck %s -check-prefixes=SSE2
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+avx2 | FileCheck %s -check-prefixes=AVX2
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+avx512f | FileCheck %s -check-prefixes=AVX512
+
+; Predicate mask-expansion (fanout) cost for masked memory ops. The mask is a
+; compact <N x i1>; once the data type splits into more register parts than the
+; mask, each extra part needs its own sub-mask materialized, so the per-lane
+; cost must rise with the width (it is flat without the fanout charge). This is
+; charged on every subtarget where the masked op is legal, not just AVX-512.
+; The mask is a variable (runtime) predicate: a constant/all-ones mask is folded
+; by CodeGen and is not charged the fanout (see getPredicateFanoutCost).
+
+define void @masked_store_f64(ptr %p, <4 x i1> %m4, <8 x i1> %m8, <16 x i1> %m16, <32 x i1> %m32) {
+; SSE2-LABEL: 'masked_store_f64'
+; SSE2-NEXT: Cost Model: Found an estimated cost of 17 for instruction: call void @llvm.masked.store.v4f64.p0(<4 x double> poison, ptr align 8 %p, <4 x i1> %m4)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 35 for instruction: call void @llvm.masked.store.v8f64.p0(<8 x double> poison, ptr align 8 %p, <8 x i1> %m8)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 71 for instruction: call void @llvm.masked.store.v16f64.p0(<16 x double> poison, ptr align 8 %p, <16 x i1> %m16)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 142 for instruction: call void @llvm.masked.store.v32f64.p0(<32 x double> poison, ptr align 8 %p, <32 x i1> %m32)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; AVX2-LABEL: 'masked_store_f64'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 8 for instruction: call void @llvm.masked.store.v4f64.p0(<4 x double> poison, ptr align 8 %p, <4 x i1> %m4)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 20 for instruction: call void @llvm.masked.store.v8f64.p0(<8 x double> poison, ptr align 8 %p, <8 x i1> %m8)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 44 for instruction: call void @llvm.masked.store.v16f64.p0(<16 x double> poison, ptr align 8 %p, <16 x i1> %m16)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 92 for instruction: call void @llvm.masked.store.v32f64.p0(<32 x double> poison, ptr align 8 %p, <32 x i1> %m32)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; AVX512-LABEL: 'masked_store_f64'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: call void @llvm.masked.store.v4f64.p0(<4 x double> poison, ptr align 8 %p, <4 x i1> %m4)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: call void @llvm.masked.store.v8f64.p0(<8 x double> poison, ptr align 8 %p, <8 x i1> %m8)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 6 for instruction: call void @llvm.masked.store.v16f64.p0(<16 x double> poison, ptr align 8 %p, <16 x i1> %m16)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 12 for instruction: call void @llvm.masked.store.v32f64.p0(<32 x double> poison, ptr align 8 %p, <32 x i1> %m32)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+ call void @llvm.masked.store.v4f64.p0(<4 x double> poison, ptr %p, i32 8, <4 x i1> %m4)
+ call void @llvm.masked.store.v8f64.p0(<8 x double> poison, ptr %p, i32 8, <8 x i1> %m8)
+ call void @llvm.masked.store.v16f64.p0(<16 x double> poison, ptr %p, i32 8, <16 x i1> %m16)
+ call void @llvm.masked.store.v32f64.p0(<32 x double> poison, ptr %p, i32 8, <32 x i1> %m32)
+ ret void
+}
+
+define void @masked_load_f64(ptr %p, <4 x i1> %m4, <8 x i1> %m8, <16 x i1> %m16) {
+; SSE2-LABEL: 'masked_load_f64'
+; SSE2-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %l4 = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 %p, <4 x i1> %m4, <4 x double> poison)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 35 for instruction: %l8 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 %p, <8 x i1> %m8, <8 x double> poison)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 71 for instruction: %l16 = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 %p, <16 x i1> %m16, <16 x double> poison)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; AVX2-LABEL: 'masked_load_f64'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %l4 = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 %p, <4 x i1> %m4, <4 x double> poison)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %l8 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 %p, <8 x i1> %m8, <8 x double> poison)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %l16 = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 %p, <16 x i1> %m16, <16 x double> poison)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; AVX512-LABEL: 'masked_load_f64'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %l4 = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 %p, <4 x i1> %m4, <4 x double> poison)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %l8 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 %p, <8 x i1> %m8, <8 x double> poison)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %l16 = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 %p, <16 x i1> %m16, <16 x double> poison)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+ %l4 = call <4 x double> @llvm.masked.load.v4f64.p0(ptr %p, i32 8, <4 x i1> %m4, <4 x double> poison)
+ %l8 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr %p, i32 8, <8 x i1> %m8, <8 x double> poison)
+ %l16 = call <16 x double> @llvm.masked.load.v16f64.p0(ptr %p, i32 8, <16 x i1> %m16, <16 x double> poison)
+ ret void
+}
+
+define void @masked_store_i64(ptr %p, <4 x i1> %m4, <8 x i1> %m8, <16 x i1> %m16) {
+; SSE2-LABEL: 'masked_store_i64'
+; SSE2-NEXT: Cost Model: Found an estimated cost of 21 for instruction: call void @llvm.masked.store.v4i64.p0(<4 x i64> poison, ptr align 8 %p, <4 x i1> %m4)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 43 for instruction: call void @llvm.masked.store.v8i64.p0(<8 x i64> poison, ptr align 8 %p, <8 x i1> %m8)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 87 for instruction: call void @llvm.masked.store.v16i64.p0(<16 x i64> poison, ptr align 8 %p, <16 x i1> %m16)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; AVX2-LABEL: 'masked_store_i64'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 8 for instruction: call void @llvm.masked.store.v4i64.p0(<4 x i64> poison, ptr align 8 %p, <4 x i1> %m4)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 20 for instruction: call void @llvm.masked.store.v8i64.p0(<8 x i64> poison, ptr align 8 %p, <8 x i1> %m8)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 44 for instruction: call void @llvm.masked.store.v16i64.p0(<16 x i64> poison, ptr align 8 %p, <16 x i1> %m16)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; AVX512-LABEL: 'masked_store_i64'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: call void @llvm.masked.store.v4i64.p0(<4 x i64> poison, ptr align 8 %p, <4 x i1> %m4)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: call void @llvm.masked.store.v8i64.p0(<8 x i64> poison, ptr align 8 %p, <8 x i1> %m8)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 6 for instruction: call void @llvm.masked.store.v16i64.p0(<16 x i64> poison, ptr align 8 %p, <16 x i1> %m16)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+ call void @llvm.masked.store.v4i64.p0(<4 x i64> poison, ptr %p, i32 8, <4 x i1> %m4)
+ call void @llvm.masked.store.v8i64.p0(<8 x i64> poison, ptr %p, i32 8, <8 x i1> %m8)
+ call void @llvm.masked.store.v16i64.p0(<16 x i64> poison, ptr %p, i32 8, <16 x i1> %m16)
+ ret void
+}
diff --git a/llvm/test/Analysis/CostModel/X86/select-mask-expansion.ll b/llvm/test/Analysis/CostModel/X86/select-mask-expansion.ll
new file mode 100644
index 0000000000000..7ae7454bef70c
--- /dev/null
+++ b/llvm/test/Analysis/CostModel/X86/select-mask-expansion.ll
@@ -0,0 +1,63 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+sse2 | FileCheck %s -check-prefixes=CHECK,SSE
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+avx2 | FileCheck %s -check-prefixes=CHECK,AVX2
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+avx512f | FileCheck %s -check-prefixes=CHECK,AVX512
+
+; A vector select whose value type legalizes into more register parts than its
+; <N x i1> condition needs extra kshiftr instructions on AVX-512 to extract a
+; sub-mask for each value part. That fanout is charged only on AVX-512, where
+; the mask lives in a single k-register; SSE/AVX keep a full-width vector mask
+; and are not charged.
+
+define <8 x i64> @sel_v8i64(<8 x i1> %m, <8 x i64> %a, <8 x i64> %b) {
+; SSE-LABEL: 'sel_v8i64'
+; SSE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %s = select <8 x i1> %m, <8 x i64> %a, <8 x i64> %b
+; SSE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i64> %s
+;
+; AVX2-LABEL: 'sel_v8i64'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %s = select <8 x i1> %m, <8 x i64> %a, <8 x i64> %b
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i64> %s
+;
+; AVX512-LABEL: 'sel_v8i64'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %s = select <8 x i1> %m, <8 x i64> %a, <8 x i64> %b
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i64> %s
+;
+ %s = select <8 x i1> %m, <8 x i64> %a, <8 x i64> %b
+ ret <8 x i64> %s
+}
+
+define <16 x i64> @sel_v16i64(<16 x i1> %m, <16 x i64> %a, <16 x i64> %b) {
+; SSE-LABEL: 'sel_v16i64'
+; SSE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %s = select <16 x i1> %m, <16 x i64> %a, <16 x i64> %b
+; SSE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <16 x i64> %s
+;
+; AVX2-LABEL: 'sel_v16i64'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %s = select <16 x i1> %m, <16 x i64> %a, <16 x i64> %b
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <16 x i64> %s
+;
+; AVX512-LABEL: 'sel_v16i64'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %s = select <16 x i1> %m, <16 x i64> %a, <16 x i64> %b
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <16 x i64> %s
+;
+ %s = select <16 x i1> %m, <16 x i64> %a, <16 x i64> %b
+ ret <16 x i64> %s
+}
+
+define <16 x i32> @sel_v16i32(<16 x i1> %m, <16 x i32> %a, <16 x i32> %b) {
+; SSE-LABEL: 'sel_v16i32'
+; SSE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %s = select <16 x i1> %m, <16 x i32> %a, <16 x i32> %b
+; SSE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <16 x i32> %s
+;
+; AVX2-LABEL: 'sel_v16i32'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %s = select <16 x i1> %m, <16 x i32> %a, <16 x i32> %b
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <16 x i32> %s
+;
+; AVX512-LABEL: 'sel_v16i32'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %s = select <16 x i1> %m, <16 x i32> %a, <16 x i32> %b
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <16 x i32> %s
+;
+ %s = select <16 x i1> %m, <16 x i32> %a, <16 x i32> %b
+ ret <16 x i32> %s
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK: {{.*}}
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
index 8f5da77027970..f4bfc6530c491 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
@@ -44,8 +44,8 @@ define void @test() {
; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 28 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 60 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
;
; AVX512-LABEL: 'test'
; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
@@ -53,8 +53,8 @@ define void @test() {
; AVX512: Cost of 17 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
; AVX512: Cost of 18 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 36 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 72 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 40 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 84 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
index 6801a549e19c4..d93596a160a60 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
@@ -43,18 +43,18 @@ define void @test() {
; AVX2-FASTGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 16 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 36 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 76 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
;
; AVX512-LABEL: 'test'
; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
; AVX512: Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
; AVX512: Cost of 18 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 52 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 108 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
index 9cb0e482da47c..cbd611958bdea 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
@@ -45,7 +45,7 @@ define void @test(ptr %B) {
; AVX512: Cost of 1 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX512: Cost of 2 for VF 64: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX512: Cost of 6 for VF 64: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
index fc42ce6e6f73f..4a726609447ce 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
@@ -27,16 +27,16 @@ define void @test(ptr %B) {
; AVX1: Cost of 3 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX1: Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX1: Cost of 2 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX1: Cost of 4 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX1: Cost of 8 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX1: Cost of 8 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX1: Cost of 20 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
; AVX2-LABEL: 'test'
; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
; AVX2: Cost of 3 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX2: Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX2: Cost of 2 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX2: Cost of 4 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX2: Cost of 8 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX2: Cost of 8 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX2: Cost of 20 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
; AVX512-LABEL: 'test'
; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
@@ -44,8 +44,8 @@ define void @test(ptr %B) {
; AVX512: Cost of 1 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX512: Cost of 2 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX512: Cost of 4 for VF 64: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX512: Cost of 6 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX512: Cost of 16 for VF 64: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
index 48c9b01beb888..ee0b1441a836b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
@@ -26,26 +26,26 @@ define void @test(ptr %B) {
; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
; AVX1: Cost of 2 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX1: Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX1: Cost of 4 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX1: Cost of 8 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX1: Cost of 16 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX1: Cost of 8 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX1: Cost of 20 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX1: Cost of 44 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
; AVX2-LABEL: 'test'
; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
; AVX2: Cost of 2 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX2: Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX2: Cost of 4 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX2: Cost of 8 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX2: Cost of 16 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX2: Cost of 8 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX2: Cost of 20 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX2: Cost of 44 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
; AVX512-LABEL: 'test'
; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
; AVX512: Cost of 1 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX512: Cost of 2 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX512: Cost of 4 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX512: Cost of 8 for VF 64: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX512: Cost of 6 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX512: Cost of 16 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX512: Cost of 36 for VF 64: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
index 6959fea2d512b..b529918e2298f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
@@ -52,8 +52,8 @@ define void @test() {
; AVX512: Cost of 10.5 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 18 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 36 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 72 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 84 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
index 41ae89933204e..10305e8a96791 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
@@ -51,9 +51,9 @@ define void @test() {
; AVX512: Cost of 5 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 11 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 52 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 108 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
index 61436a61dba50..2b4b465b8a07f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
@@ -45,7 +45,7 @@ define void @test(ptr %C) {
; AVX512: Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 2 for VF 64: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 6 for VF 64: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
index 0afea8d1664d5..c04214fc714a8 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
@@ -34,16 +34,16 @@ define void @test(ptr %C) {
; AVX1: Cost of 9 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX1: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX1: Cost of 8 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX1: Cost of 16 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX1: Cost of 32 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX1: Cost of 20 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX1: Cost of 44 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX2-LABEL: 'test'
; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; AVX2: Cost of 9 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX2: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX2: Cost of 8 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX2: Cost of 16 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX2: Cost of 32 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX2: Cost of 20 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX2: Cost of 44 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX512-LABEL: 'test'
; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
@@ -51,8 +51,8 @@ define void @test(ptr %C) {
; AVX512: Cost of 1 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 2 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 4 for VF 64: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 6 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 16 for VF 64: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
index ce2d69fca6a3b..bcda6b507f533 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
@@ -33,26 +33,26 @@ define void @test(ptr %C) {
; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX1: Cost of 8 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX1: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX1: Cost of 16 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX1: Cost of 32 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX1: Cost of 64 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX1: Cost of 20 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX1: Cost of 44 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX1: Cost of 92 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX2-LABEL: 'test'
; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX2: Cost of 8 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX2: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX2: Cost of 16 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX2: Cost of 32 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX2: Cost of 64 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX2: Cost of 20 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX2: Cost of 44 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX2: Cost of 92 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX512-LABEL: 'test'
; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX512: Cost of 1 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 2 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 4 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 8 for VF 64: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 6 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 16 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 36 for VF 64: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll b/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll
index 806a81f721212..4ee17d5a922ef 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll
@@ -813,15 +813,42 @@ define void @foo3(ptr nocapture %A, ptr nocapture readonly %B, ptr nocapture rea
; AVX2: [[VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
-; AVX2-NEXT: [[TMP1:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD]], splat (i32 100)
+; AVX2-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 4
+; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 8
+; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 12
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
+; AVX2-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4, !alias.scope [[META12]]
+; AVX2-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META12]]
+; AVX2-NEXT: [[WIDE_LOAD8:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META12]]
+; AVX2-NEXT: [[TMP4:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD]], splat (i32 100)
+; AVX2-NEXT: [[TMP5:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD6]], splat (i32 100)
+; AVX2-NEXT: [[TMP6:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD7]], splat (i32 100)
+; AVX2-NEXT: [[TMP7:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD8]], splat (i32 100)
; AVX2-NEXT: [[TMP8:%.*]] = getelementptr double, ptr [[B]], i64 [[INDEX]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP8]], <8 x i1> [[TMP1]], <8 x double> poison), !alias.scope [[META15:![0-9]+]]
-; AVX2-NEXT: [[TMP3:%.*]] = sitofp <8 x i32> [[WIDE_LOAD]] to <8 x double>
-; AVX2-NEXT: [[TMP4:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], [[TMP3]]
+; AVX2-NEXT: [[TMP9:%.*]] = getelementptr double, ptr [[TMP8]], i64 4
+; AVX2-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[TMP8]], i64 8
+; AVX2-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP8]], i64 12
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP8]], <4 x i1> [[TMP4]], <4 x double> poison), !alias.scope [[META15:![0-9]+]]
+; AVX2-NEXT: [[WIDE_MASKED_LOAD9:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP9]], <4 x i1> [[TMP5]], <4 x double> poison), !alias.scope [[META15]]
+; AVX2-NEXT: [[WIDE_MASKED_LOAD10:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP10]], <4 x i1> [[TMP6]], <4 x double> poison), !alias.scope [[META15]]
+; AVX2-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP11]], <4 x i1> [[TMP7]], <4 x double> poison), !alias.scope [[META15]]
+; AVX2-NEXT: [[TMP12:%.*]] = sitofp <4 x i32> [[WIDE_LOAD]] to <4 x double>
+; AVX2-NEXT: [[TMP13:%.*]] = sitofp <4 x i32> [[WIDE_LOAD6]] to <4 x double>
+; AVX2-NEXT: [[TMP14:%.*]] = sitofp <4 x i32> [[WIDE_LOAD7]] to <4 x double>
+; AVX2-NEXT: [[TMP15:%.*]] = sitofp <4 x i32> [[WIDE_LOAD8]] to <4 x double>
+; AVX2-NEXT: [[TMP16:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD]], [[TMP12]]
+; AVX2-NEXT: [[TMP17:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD9]], [[TMP13]]
+; AVX2-NEXT: [[TMP18:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD10]], [[TMP14]]
+; AVX2-NEXT: [[TMP19:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD11]], [[TMP15]]
; AVX2-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[A]], i64 [[INDEX]]
-; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP4]], ptr align 8 [[TMP20]], <8 x i1> [[TMP1]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; AVX2-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP20]], i64 4
+; AVX2-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP20]], i64 8
+; AVX2-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP20]], i64 12
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP16]], ptr align 8 [[TMP20]], <4 x i1> [[TMP4]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP17]], ptr align 8 [[TMP21]], <4 x i1> [[TMP5]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP18]], ptr align 8 [[TMP22]], <4 x i1> [[TMP6]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP19]], ptr align 8 [[TMP23]], <4 x i1> [[TMP7]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX2-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 10000
; AVX2-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
@@ -851,65 +878,65 @@ define void @foo3(ptr nocapture %A, ptr nocapture readonly %B, ptr nocapture rea
; AVX512: [[VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[INDEX]]
+; AVX512-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 8
; AVX512-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 16
-; AVX512-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 32
-; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 48
-; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
-; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <16 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <16 x i32>, ptr [[TMP9]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <16 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[TMP4:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD]], splat (i32 100)
-; AVX512-NEXT: [[TMP5:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD6]], splat (i32 100)
-; AVX512-NEXT: [[TMP6:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD7]], splat (i32 100)
-; AVX512-NEXT: [[TMP7:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD8]], splat (i32 100)
+; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 24
+; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
+; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <8 x i32>, ptr [[TMP1]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <8 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[TMP4:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD]], splat (i32 100)
+; AVX512-NEXT: [[TMP5:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD6]], splat (i32 100)
+; AVX512-NEXT: [[TMP6:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD7]], splat (i32 100)
+; AVX512-NEXT: [[TMP7:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD8]], splat (i32 100)
; AVX512-NEXT: [[TMP8:%.*]] = getelementptr double, ptr [[B]], i64 [[INDEX]]
+; AVX512-NEXT: [[TMP9:%.*]] = getelementptr double, ptr [[TMP8]], i64 8
; AVX512-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[TMP8]], i64 16
-; AVX512-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP8]], i64 32
-; AVX512-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP8]], i64 48
-; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP8]], <16 x i1> [[TMP4]], <16 x double> poison), !alias.scope [[META15:![0-9]+]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD9:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP10]], <16 x i1> [[TMP5]], <16 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD10:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP21]], <16 x i1> [[TMP6]], <16 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP11]], <16 x i1> [[TMP7]], <16 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[TMP12:%.*]] = sitofp <16 x i32> [[WIDE_LOAD]] to <16 x double>
-; AVX512-NEXT: [[TMP13:%.*]] = sitofp <16 x i32> [[WIDE_LOAD6]] to <16 x double>
-; AVX512-NEXT: [[TMP14:%.*]] = sitofp <16 x i32> [[WIDE_LOAD7]] to <16 x double>
-; AVX512-NEXT: [[TMP15:%.*]] = sitofp <16 x i32> [[WIDE_LOAD8]] to <16 x double>
-; AVX512-NEXT: [[TMP16:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD]], [[TMP12]]
-; AVX512-NEXT: [[TMP17:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD9]], [[TMP13]]
-; AVX512-NEXT: [[TMP18:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD10]], [[TMP14]]
-; AVX512-NEXT: [[TMP19:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD11]], [[TMP15]]
+; AVX512-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP8]], i64 24
+; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP8]], <8 x i1> [[TMP4]], <8 x double> poison), !alias.scope [[META15:![0-9]+]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD9:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP9]], <8 x i1> [[TMP5]], <8 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD10:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP10]], <8 x i1> [[TMP6]], <8 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP11]], <8 x i1> [[TMP7]], <8 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[TMP12:%.*]] = sitofp <8 x i32> [[WIDE_LOAD]] to <8 x double>
+; AVX512-NEXT: [[TMP13:%.*]] = sitofp <8 x i32> [[WIDE_LOAD6]] to <8 x double>
+; AVX512-NEXT: [[TMP14:%.*]] = sitofp <8 x i32> [[WIDE_LOAD7]] to <8 x double>
+; AVX512-NEXT: [[TMP15:%.*]] = sitofp <8 x i32> [[WIDE_LOAD8]] to <8 x double>
+; AVX512-NEXT: [[TMP16:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], [[TMP12]]
+; AVX512-NEXT: [[TMP17:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD9]], [[TMP13]]
+; AVX512-NEXT: [[TMP18:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD10]], [[TMP14]]
+; AVX512-NEXT: [[TMP19:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD11]], [[TMP15]]
; AVX512-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[A]], i64 [[INDEX]]
+; AVX512-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP20]], i64 8
; AVX512-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP20]], i64 16
-; AVX512-NEXT: [[TMP32:%.*]] = getelementptr double, ptr [[TMP20]], i64 32
-; AVX512-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP20]], i64 48
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP16]], ptr align 8 [[TMP20]], <16 x i1> [[TMP4]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP17]], ptr align 8 [[TMP22]], <16 x i1> [[TMP5]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP18]], ptr align 8 [[TMP32]], <16 x i1> [[TMP6]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP19]], ptr align 8 [[TMP23]], <16 x i1> [[TMP7]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
+; AVX512-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP20]], i64 24
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP16]], ptr align 8 [[TMP20]], <8 x i1> [[TMP4]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP17]], ptr align 8 [[TMP21]], <8 x i1> [[TMP5]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP18]], ptr align 8 [[TMP22]], <8 x i1> [[TMP6]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP19]], ptr align 8 [[TMP23]], <8 x i1> [[TMP7]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; AVX512-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 9984
; AVX512-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: br i1 false, [[FOR_END:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX512: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX512-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
+; AVX512-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF21:![0-9]+]]
; AVX512: [[VEC_EPILOG_PH]]:
; AVX512-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ 9984, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX512-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX512: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX12:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT15:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP25:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[INDEX12]]
-; AVX512-NEXT: [[WIDE_LOAD13:%.*]] = load <16 x i32>, ptr [[TMP25]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[TMP26:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD13]], splat (i32 100)
+; AVX512-NEXT: [[WIDE_LOAD13:%.*]] = load <8 x i32>, ptr [[TMP25]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[TMP26:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD13]], splat (i32 100)
; AVX512-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[B]], i64 [[INDEX12]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD14:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP27]], <16 x i1> [[TMP26]], <16 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[TMP28:%.*]] = sitofp <16 x i32> [[WIDE_LOAD13]] to <16 x double>
-; AVX512-NEXT: [[TMP29:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD14]], [[TMP28]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD14:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP27]], <8 x i1> [[TMP26]], <8 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[TMP28:%.*]] = sitofp <8 x i32> [[WIDE_LOAD13]] to <8 x double>
+; AVX512-NEXT: [[TMP29:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD14]], [[TMP28]]
; AVX512-NEXT: [[TMP30:%.*]] = getelementptr double, ptr [[A]], i64 [[INDEX12]]
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP29]], ptr align 8 [[TMP30]], <16 x i1> [[TMP26]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: [[INDEX_NEXT15]] = add nuw i64 [[INDEX12]], 16
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP29]], ptr align 8 [[TMP30]], <8 x i1> [[TMP26]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: [[INDEX_NEXT15]] = add nuw i64 [[INDEX12]], 8
; AVX512-NEXT: [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT15]], 10000
-; AVX512-NEXT: br i1 [[TMP31]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP21:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP31]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP22:![0-9]+]]
; AVX512: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX512-NEXT: br i1 true, [[FOR_END]], label %[[VEC_EPILOG_SCALAR_PH]]
; AVX512: [[VEC_EPILOG_SCALAR_PH]]:
@@ -1000,21 +1027,21 @@ define void @foo4(ptr nocapture %A, ptr nocapture readonly %B, ptr nocapture rea
; AVX512-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX512: [[VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; AVX512-NEXT: [[VEC_IND:%.*]] = phi <16 x i64> [ <i64 0, i64 16, i64 32, i64 48, i64 64, i64 80, i64 96, i64 112, i64 128, i64 144, i64 160, i64 176, i64 192, i64 208, i64 224, i64 240>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; AVX512-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], <16 x i64> [[VEC_IND]]
-; AVX512-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <16 x i32> @llvm.masked.gather.v16i32.v16p0(<16 x ptr> align 4 [[WIDE_GEP]], <16 x i1> splat (i1 true), <16 x i32> poison), !alias.scope [[META23:![0-9]+]]
-; AVX512-NEXT: [[TMP0:%.*]] = icmp slt <16 x i32> [[WIDE_MASKED_GATHER]], splat (i32 100)
-; AVX512-NEXT: [[TMP1:%.*]] = shl nuw nsw <16 x i64> [[VEC_IND]], splat (i64 1)
-; AVX512-NEXT: [[WIDE_GEP6:%.*]] = getelementptr inbounds double, ptr [[B]], <16 x i64> [[TMP1]]
-; AVX512-NEXT: [[WIDE_MASKED_GATHER7:%.*]] = call <16 x double> @llvm.masked.gather.v16f64.v16p0(<16 x ptr> align 8 [[WIDE_GEP6]], <16 x i1> [[TMP0]], <16 x double> poison), !alias.scope [[META26:![0-9]+]]
-; AVX512-NEXT: [[TMP2:%.*]] = sitofp <16 x i32> [[WIDE_MASKED_GATHER]] to <16 x double>
-; AVX512-NEXT: [[TMP3:%.*]] = fadd <16 x double> [[WIDE_MASKED_GATHER7]], [[TMP2]]
-; AVX512-NEXT: [[WIDE_GEP8:%.*]] = getelementptr inbounds double, ptr [[A]], <16 x i64> [[VEC_IND]]
-; AVX512-NEXT: call void @llvm.masked.scatter.v16f64.v16p0(<16 x double> [[TMP3]], <16 x ptr> align 8 [[WIDE_GEP8]], <16 x i1> [[TMP0]]), !alias.scope [[META28:![0-9]+]], !noalias [[META30:![0-9]+]]
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
-; AVX512-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <16 x i64> [[VEC_IND]], splat (i64 256)
+; AVX512-NEXT: [[VEC_IND:%.*]] = phi <8 x i64> [ <i64 0, i64 16, i64 32, i64 48, i64 64, i64 80, i64 96, i64 112>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; AVX512-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], <8 x i64> [[VEC_IND]]
+; AVX512-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <8 x i32> @llvm.masked.gather.v8i32.v8p0(<8 x ptr> align 4 [[WIDE_GEP]], <8 x i1> splat (i1 true), <8 x i32> poison), !alias.scope [[META24:![0-9]+]]
+; AVX512-NEXT: [[TMP0:%.*]] = icmp slt <8 x i32> [[WIDE_MASKED_GATHER]], splat (i32 100)
+; AVX512-NEXT: [[TMP1:%.*]] = shl nuw nsw <8 x i64> [[VEC_IND]], splat (i64 1)
+; AVX512-NEXT: [[WIDE_GEP6:%.*]] = getelementptr inbounds double, ptr [[B]], <8 x i64> [[TMP1]]
+; AVX512-NEXT: [[WIDE_MASKED_GATHER7:%.*]] = call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 8 [[WIDE_GEP6]], <8 x i1> [[TMP0]], <8 x double> poison), !alias.scope [[META27:![0-9]+]]
+; AVX512-NEXT: [[TMP2:%.*]] = sitofp <8 x i32> [[WIDE_MASKED_GATHER]] to <8 x double>
+; AVX512-NEXT: [[TMP3:%.*]] = fadd <8 x double> [[WIDE_MASKED_GATHER7]], [[TMP2]]
+; AVX512-NEXT: [[WIDE_GEP8:%.*]] = getelementptr inbounds double, ptr [[A]], <8 x i64> [[VEC_IND]]
+; AVX512-NEXT: call void @llvm.masked.scatter.v8f64.v8p0(<8 x double> [[TMP3]], <8 x ptr> align 8 [[WIDE_GEP8]], <8 x i1> [[TMP0]]), !alias.scope [[META29:![0-9]+]], !noalias [[META31:![0-9]+]]
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; AVX512-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <8 x i64> [[VEC_IND]], splat (i64 128)
; AVX512-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], 624
-; AVX512-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP31:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: br label %[[SCALAR_PH]]
; AVX512: [[SCALAR_PH]]:
@@ -1084,19 +1111,49 @@ define void @foo6(ptr nocapture readonly %in, ptr nocapture %out, i32 %size, ptr
; AVX1-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP0:%.*]] = sub i64 4095, [[INDEX]]
; AVX1-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[TMP0]]
+; AVX1-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -3
; AVX1-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -7
-; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META18:![0-9]+]]
-; AVX1-NEXT: [[REVERSE:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[TMP4:%.*]] = icmp sgt <8 x i32> [[REVERSE]], zeroinitializer
+; AVX1-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -11
+; AVX1-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -15
+; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META18:![0-9]+]]
+; AVX1-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META18]]
+; AVX1-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META18]]
+; AVX1-NEXT: [[WIDE_LOAD8:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META18]]
+; AVX1-NEXT: [[REVERSE:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[REVERSE9:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD6]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[REVERSE10:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD7]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[REVERSE11:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD8]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[TMP6:%.*]] = icmp sgt <4 x i32> [[REVERSE]], zeroinitializer
+; AVX1-NEXT: [[TMP7:%.*]] = icmp sgt <4 x i32> [[REVERSE9]], zeroinitializer
+; AVX1-NEXT: [[TMP8:%.*]] = icmp sgt <4 x i32> [[REVERSE10]], zeroinitializer
+; AVX1-NEXT: [[TMP9:%.*]] = icmp sgt <4 x i32> [[REVERSE11]], zeroinitializer
; AVX1-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[IN]], i64 [[TMP0]]
+; AVX1-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP10]], i64 -3
; AVX1-NEXT: [[TMP12:%.*]] = getelementptr double, ptr [[TMP10]], i64 -7
-; AVX1-NEXT: [[REVERSE6:%.*]] = shufflevector <8 x i1> [[TMP4]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP12]], <8 x i1> [[REVERSE6]], <8 x double> poison), !alias.scope [[META21:![0-9]+]]
-; AVX1-NEXT: [[TMP6:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
+; AVX1-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -11
+; AVX1-NEXT: [[TMP14:%.*]] = getelementptr double, ptr [[TMP10]], i64 -15
+; AVX1-NEXT: [[REVERSE12:%.*]] = shufflevector <4 x i1> [[TMP6]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[REVERSE13:%.*]] = shufflevector <4 x i1> [[TMP7]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[REVERSE14:%.*]] = shufflevector <4 x i1> [[TMP8]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[REVERSE15:%.*]] = shufflevector <4 x i1> [[TMP9]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP11]], <4 x i1> [[REVERSE12]], <4 x double> poison), !alias.scope [[META21:![0-9]+]]
+; AVX1-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP12]], <4 x i1> [[REVERSE13]], <4 x double> poison), !alias.scope [[META21]]
+; AVX1-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP13]], <4 x i1> [[REVERSE14]], <4 x double> poison), !alias.scope [[META21]]
+; AVX1-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP14]], <4 x i1> [[REVERSE15]], <4 x double> poison), !alias.scope [[META21]]
+; AVX1-NEXT: [[TMP15:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
+; AVX1-NEXT: [[TMP16:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
+; AVX1-NEXT: [[TMP17:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
+; AVX1-NEXT: [[TMP18:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
; AVX1-NEXT: [[TMP19:%.*]] = getelementptr double, ptr [[OUT]], i64 [[TMP0]]
+; AVX1-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP19]], i64 -3
; AVX1-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP19]], i64 -7
-; AVX1-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP6]], ptr align 8 [[TMP21]], <8 x i1> [[REVERSE6]]), !alias.scope [[META23:![0-9]+]], !noalias [[META25:![0-9]+]]
-; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; AVX1-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -11
+; AVX1-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP19]], i64 -15
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP15]], ptr align 8 [[TMP20]], <4 x i1> [[REVERSE12]]), !alias.scope [[META23:![0-9]+]], !noalias [[META25:![0-9]+]]
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP16]], ptr align 8 [[TMP21]], <4 x i1> [[REVERSE13]]), !alias.scope [[META23]], !noalias [[META25]]
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP17]], ptr align 8 [[TMP22]], <4 x i1> [[REVERSE14]]), !alias.scope [[META23]], !noalias [[META25]]
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP18]], ptr align 8 [[TMP23]], <4 x i1> [[REVERSE15]]), !alias.scope [[META23]], !noalias [[META25]]
+; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX1-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4096
; AVX1-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP26:![0-9]+]]
; AVX1: [[MIDDLE_BLOCK]]:
@@ -1125,19 +1182,49 @@ define void @foo6(ptr nocapture readonly %in, ptr nocapture %out, i32 %size, ptr
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = sub i64 4095, [[INDEX]]
; AVX2-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[TMP0]]
+; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -3
; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -7
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META22:![0-9]+]]
-; AVX2-NEXT: [[REVERSE:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[TMP4:%.*]] = icmp sgt <8 x i32> [[REVERSE]], zeroinitializer
+; AVX2-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -11
+; AVX2-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -15
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META22:![0-9]+]]
+; AVX2-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META22]]
+; AVX2-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META22]]
+; AVX2-NEXT: [[WIDE_LOAD8:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META22]]
+; AVX2-NEXT: [[REVERSE:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[REVERSE9:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD6]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[REVERSE10:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD7]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[REVERSE11:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD8]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[TMP6:%.*]] = icmp sgt <4 x i32> [[REVERSE]], zeroinitializer
+; AVX2-NEXT: [[TMP7:%.*]] = icmp sgt <4 x i32> [[REVERSE9]], zeroinitializer
+; AVX2-NEXT: [[TMP8:%.*]] = icmp sgt <4 x i32> [[REVERSE10]], zeroinitializer
+; AVX2-NEXT: [[TMP9:%.*]] = icmp sgt <4 x i32> [[REVERSE11]], zeroinitializer
; AVX2-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[IN]], i64 [[TMP0]]
+; AVX2-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP10]], i64 -3
; AVX2-NEXT: [[TMP12:%.*]] = getelementptr double, ptr [[TMP10]], i64 -7
-; AVX2-NEXT: [[REVERSE6:%.*]] = shufflevector <8 x i1> [[TMP4]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP12]], <8 x i1> [[REVERSE6]], <8 x double> poison), !alias.scope [[META25:![0-9]+]]
-; AVX2-NEXT: [[TMP6:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
+; AVX2-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -11
+; AVX2-NEXT: [[TMP14:%.*]] = getelementptr double, ptr [[TMP10]], i64 -15
+; AVX2-NEXT: [[REVERSE12:%.*]] = shufflevector <4 x i1> [[TMP6]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[REVERSE13:%.*]] = shufflevector <4 x i1> [[TMP7]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[REVERSE14:%.*]] = shufflevector <4 x i1> [[TMP8]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[REVERSE15:%.*]] = shufflevector <4 x i1> [[TMP9]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP11]], <4 x i1> [[REVERSE12]], <4 x double> poison), !alias.scope [[META25:![0-9]+]]
+; AVX2-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP12]], <4 x i1> [[REVERSE13]], <4 x double> poison), !alias.scope [[META25]]
+; AVX2-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP13]], <4 x i1> [[REVERSE14]], <4 x double> poison), !alias.scope [[META25]]
+; AVX2-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP14]], <4 x i1> [[REVERSE15]], <4 x double> poison), !alias.scope [[META25]]
+; AVX2-NEXT: [[TMP15:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
+; AVX2-NEXT: [[TMP16:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
+; AVX2-NEXT: [[TMP17:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
+; AVX2-NEXT: [[TMP18:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
; AVX2-NEXT: [[TMP19:%.*]] = getelementptr double, ptr [[OUT]], i64 [[TMP0]]
+; AVX2-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP19]], i64 -3
; AVX2-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP19]], i64 -7
-; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP6]], ptr align 8 [[TMP21]], <8 x i1> [[REVERSE6]]), !alias.scope [[META27:![0-9]+]], !noalias [[META29:![0-9]+]]
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; AVX2-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -11
+; AVX2-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP19]], i64 -15
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP15]], ptr align 8 [[TMP20]], <4 x i1> [[REVERSE12]]), !alias.scope [[META27:![0-9]+]], !noalias [[META29:![0-9]+]]
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP16]], ptr align 8 [[TMP21]], <4 x i1> [[REVERSE13]]), !alias.scope [[META27]], !noalias [[META29]]
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP17]], ptr align 8 [[TMP22]], <4 x i1> [[REVERSE14]]), !alias.scope [[META27]], !noalias [[META29]]
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP18]], ptr align 8 [[TMP23]], <4 x i1> [[REVERSE15]]), !alias.scope [[META27]], !noalias [[META29]]
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX2-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4096
; AVX2-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP30:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
@@ -1166,51 +1253,51 @@ define void @foo6(ptr nocapture readonly %in, ptr nocapture %out, i32 %size, ptr
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP0:%.*]] = sub i64 4095, [[INDEX]]
; AVX512-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[TMP0]]
+; AVX512-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -7
; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -15
+; AVX512-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -23
; AVX512-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -31
-; AVX512-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -47
-; AVX512-NEXT: [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -63
-; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META33:![0-9]+]]
-; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <16 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META33]]
-; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <16 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META33]]
-; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <16 x i32>, ptr [[TMP11]], align 4, !alias.scope [[META33]]
-; AVX512-NEXT: [[REVERSE:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE9:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD6]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE10:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD7]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE11:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD8]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[TMP6:%.*]] = icmp sgt <16 x i32> [[REVERSE]], zeroinitializer
-; AVX512-NEXT: [[TMP7:%.*]] = icmp sgt <16 x i32> [[REVERSE9]], zeroinitializer
-; AVX512-NEXT: [[TMP8:%.*]] = icmp sgt <16 x i32> [[REVERSE10]], zeroinitializer
-; AVX512-NEXT: [[TMP9:%.*]] = icmp sgt <16 x i32> [[REVERSE11]], zeroinitializer
+; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META34:![0-9]+]]
+; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META34]]
+; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <8 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META34]]
+; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <8 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META34]]
+; AVX512-NEXT: [[REVERSE:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE9:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD6]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE10:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD7]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE11:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD8]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[TMP6:%.*]] = icmp sgt <8 x i32> [[REVERSE]], zeroinitializer
+; AVX512-NEXT: [[TMP7:%.*]] = icmp sgt <8 x i32> [[REVERSE9]], zeroinitializer
+; AVX512-NEXT: [[TMP8:%.*]] = icmp sgt <8 x i32> [[REVERSE10]], zeroinitializer
+; AVX512-NEXT: [[TMP9:%.*]] = icmp sgt <8 x i32> [[REVERSE11]], zeroinitializer
; AVX512-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[IN]], i64 [[TMP0]]
+; AVX512-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP10]], i64 -7
; AVX512-NEXT: [[TMP12:%.*]] = getelementptr double, ptr [[TMP10]], i64 -15
+; AVX512-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -23
; AVX512-NEXT: [[TMP14:%.*]] = getelementptr double, ptr [[TMP10]], i64 -31
-; AVX512-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -47
-; AVX512-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP10]], i64 -63
-; AVX512-NEXT: [[REVERSE12:%.*]] = shufflevector <16 x i1> [[TMP6]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE13:%.*]] = shufflevector <16 x i1> [[TMP7]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE14:%.*]] = shufflevector <16 x i1> [[TMP8]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE15:%.*]] = shufflevector <16 x i1> [[TMP9]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP12]], <16 x i1> [[REVERSE12]], <16 x double> poison), !alias.scope [[META36:![0-9]+]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP14]], <16 x i1> [[REVERSE13]], <16 x double> poison), !alias.scope [[META36]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP13]], <16 x i1> [[REVERSE14]], <16 x double> poison), !alias.scope [[META36]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP20]], <16 x i1> [[REVERSE15]], <16 x double> poison), !alias.scope [[META36]]
-; AVX512-NEXT: [[TMP15:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
-; AVX512-NEXT: [[TMP16:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
-; AVX512-NEXT: [[TMP17:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
-; AVX512-NEXT: [[TMP18:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[REVERSE12:%.*]] = shufflevector <8 x i1> [[TMP6]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE13:%.*]] = shufflevector <8 x i1> [[TMP7]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE14:%.*]] = shufflevector <8 x i1> [[TMP8]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE15:%.*]] = shufflevector <8 x i1> [[TMP9]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP11]], <8 x i1> [[REVERSE12]], <8 x double> poison), !alias.scope [[META37:![0-9]+]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP12]], <8 x i1> [[REVERSE13]], <8 x double> poison), !alias.scope [[META37]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP13]], <8 x i1> [[REVERSE14]], <8 x double> poison), !alias.scope [[META37]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP14]], <8 x i1> [[REVERSE15]], <8 x double> poison), !alias.scope [[META37]]
+; AVX512-NEXT: [[TMP15:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[TMP16:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[TMP17:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[TMP18:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
; AVX512-NEXT: [[TMP19:%.*]] = getelementptr double, ptr [[OUT]], i64 [[TMP0]]
+; AVX512-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP19]], i64 -7
; AVX512-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP19]], i64 -15
+; AVX512-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -23
; AVX512-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP19]], i64 -31
-; AVX512-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -47
-; AVX512-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP19]], i64 -63
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP15]], ptr align 8 [[TMP21]], <16 x i1> [[REVERSE12]]), !alias.scope [[META38:![0-9]+]], !noalias [[META40:![0-9]+]]
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP16]], ptr align 8 [[TMP23]], <16 x i1> [[REVERSE13]]), !alias.scope [[META38]], !noalias [[META40]]
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP17]], ptr align 8 [[TMP22]], <16 x i1> [[REVERSE14]]), !alias.scope [[META38]], !noalias [[META40]]
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP18]], ptr align 8 [[TMP25]], <16 x i1> [[REVERSE15]]), !alias.scope [[META38]], !noalias [[META40]]
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP15]], ptr align 8 [[TMP20]], <8 x i1> [[REVERSE12]]), !alias.scope [[META39:![0-9]+]], !noalias [[META41:![0-9]+]]
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP16]], ptr align 8 [[TMP21]], <8 x i1> [[REVERSE13]]), !alias.scope [[META39]], !noalias [[META41]]
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP17]], ptr align 8 [[TMP22]], <8 x i1> [[REVERSE14]]), !alias.scope [[META39]], !noalias [[META41]]
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP18]], ptr align 8 [[TMP23]], <8 x i1> [[REVERSE15]]), !alias.scope [[META39]], !noalias [[META41]]
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; AVX512-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4096
-; AVX512-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP41:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP42:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: br [[FOR_END:label %.*]]
; AVX512: [[SCALAR_PH]]:
@@ -1258,54 +1345,84 @@ define void @foo7(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX1-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX1: [[ITER_CHECK]]:
; AVX1-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
+; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX1: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
+; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX1: [[VECTOR_PH]]:
-; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
+; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
; AVX1-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX1-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX1: [[VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
-; AVX1-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX1-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
+; AVX1-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
+; AVX1-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
+; AVX1-NEXT: [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
+; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
+; AVX1-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
+; AVX1-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
+; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP4]], align 1
+; AVX1-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX1-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
+; AVX1-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
+; AVX1-NEXT: [[TMP8:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX1-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
+; AVX1-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
+; AVX1-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX1-NEXT: [[TMP13:%.*]] = icmp ne <4 x i8> [[TMP8]], zeroinitializer
; AVX1-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
-; AVX1-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX1-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
+; AVX1-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
+; AVX1-NEXT: [[TMP16:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
+; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP9]], <4 x ptr> poison)
+; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
+; AVX1-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
+; AVX1-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP16]], <4 x i1> [[TMP13]], <4 x ptr> poison)
+; AVX1-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX1-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX1-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
+; AVX1-NEXT: [[TMP20:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
+; AVX1-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP25:%.*]] = select <4 x i1> [[TMP13]], <4 x i1> [[TMP20]], <4 x i1> zeroinitializer
; AVX1-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX1-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
-; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX1-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
+; AVX1-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
+; AVX1-NEXT: [[TMP30:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP21]])
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP30]], <4 x i1> [[TMP25]])
+; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX1-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP28:![0-9]+]]
; AVX1: [[MIDDLE_BLOCK]]:
; AVX1-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX1: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
+; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
; AVX1-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF29:![0-9]+]]
; AVX1: [[VEC_EPILOG_PH]]:
; AVX1-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
+; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
; AVX1-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX1-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX1: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
-; AVX1-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX1-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
+; AVX1-NEXT: [[WIDE_LOAD10:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
+; AVX1-NEXT: [[TMP34:%.*]] = and <4 x i8> [[WIDE_LOAD10]], splat (i8 1)
+; AVX1-NEXT: [[TMP33:%.*]] = icmp ne <4 x i8> [[TMP34]], zeroinitializer
; AVX1-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
-; AVX1-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX1-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
+; AVX1-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP33]], <4 x ptr> poison)
+; AVX1-NEXT: [[TMP37:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD11]], splat (ptr null)
+; AVX1-NEXT: [[TMP38:%.*]] = select <4 x i1> [[TMP33]], <4 x i1> [[TMP37]], <4 x i1> zeroinitializer
; AVX1-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX1-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
-; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP38]])
+; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
; AVX1-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
; AVX1-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP30:![0-9]+]]
; AVX1: [[VEC_EPILOG_MIDDLE_BLOCK]]:
@@ -1320,56 +1437,86 @@ define void @foo7(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX2-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX2: [[ITER_CHECK]]:
; AVX2-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
+; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX2: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
+; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX2: [[VECTOR_PH]]:
-; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
+; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
; AVX2-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX2-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX2: [[VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
-; AVX2-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX2-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
+; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
+; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
+; AVX2-NEXT: [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
+; AVX2-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
+; AVX2-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
+; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP4]], align 1
+; AVX2-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX2-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
+; AVX2-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
+; AVX2-NEXT: [[TMP8:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX2-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
+; AVX2-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
+; AVX2-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX2-NEXT: [[TMP13:%.*]] = icmp ne <4 x i8> [[TMP8]], zeroinitializer
; AVX2-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
-; AVX2-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX2-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
+; AVX2-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
+; AVX2-NEXT: [[TMP16:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP9]], <4 x ptr> poison)
+; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
+; AVX2-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
+; AVX2-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP16]], <4 x i1> [[TMP13]], <4 x ptr> poison)
+; AVX2-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX2-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX2-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
+; AVX2-NEXT: [[TMP20:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
+; AVX2-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP25:%.*]] = select <4 x i1> [[TMP13]], <4 x i1> [[TMP20]], <4 x i1> zeroinitializer
; AVX2-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX2-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX2-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
+; AVX2-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
+; AVX2-NEXT: [[TMP30:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP21]])
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP30]], <4 x i1> [[TMP25]])
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX2-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX2-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX2-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX2: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
-; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
+; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
+; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF33:![0-9]+]]
; AVX2: [[VEC_EPILOG_PH]]:
; AVX2-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
+; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
; AVX2-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX2-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX2: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
-; AVX2-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX2-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
+; AVX2-NEXT: [[WIDE_LOAD10:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
+; AVX2-NEXT: [[TMP34:%.*]] = and <4 x i8> [[WIDE_LOAD10]], splat (i8 1)
+; AVX2-NEXT: [[TMP33:%.*]] = icmp ne <4 x i8> [[TMP34]], zeroinitializer
; AVX2-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
-; AVX2-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX2-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
+; AVX2-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP33]], <4 x ptr> poison)
+; AVX2-NEXT: [[TMP37:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD11]], splat (ptr null)
+; AVX2-NEXT: [[TMP38:%.*]] = select <4 x i1> [[TMP33]], <4 x i1> [[TMP37]], <4 x i1> zeroinitializer
; AVX2-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
-; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP38]])
+; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
; AVX2-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
-; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP34:![0-9]+]]
; AVX2: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N14:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC9]]
; AVX2-NEXT: br i1 [[CMP_N14]], [[FOR_END_LOOPEXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
@@ -1405,13 +1552,13 @@ define void @foo7(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX512-NEXT: call void @llvm.masked.store.v64f64.p0(<64 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <64 x i1> [[TMP6]])
; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
; AVX512-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; AVX512-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP43:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP44:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX512-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX512: [[VEC_EPILOG_ITER_CHECK]]:
; AVX512-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
-; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF44:![0-9]+]]
+; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF45:![0-9]+]]
; AVX512: [[VEC_EPILOG_PH]]:
; AVX512-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX512-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
@@ -1431,7 +1578,7 @@ define void @foo7(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP34]])
; AVX512-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
; AVX512-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
-; AVX512-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP45:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP46:![0-9]+]]
; AVX512: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX512-NEXT: [[CMP_N14:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC9]]
; AVX512-NEXT: br i1 [[CMP_N14]], [[FOR_END_LOOPEXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
@@ -1489,54 +1636,84 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX1-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX1: [[ITER_CHECK]]:
; AVX1-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
+; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX1: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
+; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX1: [[VECTOR_PH]]:
-; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
+; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
; AVX1-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX1-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX1: [[VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
-; AVX1-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX1-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
+; AVX1-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
+; AVX1-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
+; AVX1-NEXT: [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
+; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
+; AVX1-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
+; AVX1-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
+; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP4]], align 1
+; AVX1-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX1-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
+; AVX1-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
+; AVX1-NEXT: [[TMP8:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX1-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
+; AVX1-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
+; AVX1-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX1-NEXT: [[TMP13:%.*]] = icmp ne <4 x i8> [[TMP8]], zeroinitializer
; AVX1-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
-; AVX1-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX1-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
+; AVX1-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
+; AVX1-NEXT: [[TMP16:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
+; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP9]], <4 x ptr> poison)
+; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
+; AVX1-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
+; AVX1-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP16]], <4 x i1> [[TMP13]], <4 x ptr> poison)
+; AVX1-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX1-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX1-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
+; AVX1-NEXT: [[TMP20:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
+; AVX1-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP25:%.*]] = select <4 x i1> [[TMP13]], <4 x i1> [[TMP20]], <4 x i1> zeroinitializer
; AVX1-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX1-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
-; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX1-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
+; AVX1-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
+; AVX1-NEXT: [[TMP30:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP21]])
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP30]], <4 x i1> [[TMP25]])
+; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX1-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
; AVX1: [[MIDDLE_BLOCK]]:
; AVX1-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX1: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
+; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
; AVX1-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF29]]
; AVX1: [[VEC_EPILOG_PH]]:
; AVX1-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
+; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
; AVX1-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX1-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX1: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
-; AVX1-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX1-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
+; AVX1-NEXT: [[WIDE_LOAD10:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
+; AVX1-NEXT: [[TMP34:%.*]] = and <4 x i8> [[WIDE_LOAD10]], splat (i8 1)
+; AVX1-NEXT: [[TMP33:%.*]] = icmp ne <4 x i8> [[TMP34]], zeroinitializer
; AVX1-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
-; AVX1-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX1-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
+; AVX1-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP33]], <4 x ptr> poison)
+; AVX1-NEXT: [[TMP37:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD11]], splat (ptr null)
+; AVX1-NEXT: [[TMP38:%.*]] = select <4 x i1> [[TMP33]], <4 x i1> [[TMP37]], <4 x i1> zeroinitializer
; AVX1-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX1-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
-; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP38]])
+; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
; AVX1-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
; AVX1-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
; AVX1: [[VEC_EPILOG_MIDDLE_BLOCK]]:
@@ -1551,56 +1728,86 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX2-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX2: [[ITER_CHECK]]:
; AVX2-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
+; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX2: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
+; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX2: [[VECTOR_PH]]:
-; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
+; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
; AVX2-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX2-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX2: [[VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
-; AVX2-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX2-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
+; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
+; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
+; AVX2-NEXT: [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
+; AVX2-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
+; AVX2-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
+; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP4]], align 1
+; AVX2-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX2-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
+; AVX2-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
+; AVX2-NEXT: [[TMP8:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX2-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
+; AVX2-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
+; AVX2-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX2-NEXT: [[TMP13:%.*]] = icmp ne <4 x i8> [[TMP8]], zeroinitializer
; AVX2-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
-; AVX2-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX2-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
+; AVX2-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
+; AVX2-NEXT: [[TMP16:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP9]], <4 x ptr> poison)
+; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
+; AVX2-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
+; AVX2-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP16]], <4 x i1> [[TMP13]], <4 x ptr> poison)
+; AVX2-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX2-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX2-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
+; AVX2-NEXT: [[TMP20:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
+; AVX2-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP25:%.*]] = select <4 x i1> [[TMP13]], <4 x i1> [[TMP20]], <4 x i1> zeroinitializer
; AVX2-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX2-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX2-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
+; AVX2-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
+; AVX2-NEXT: [[TMP30:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP21]])
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP30]], <4 x i1> [[TMP25]])
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX2-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; AVX2-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP35:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX2-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX2: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
-; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
+; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
+; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF33]]
; AVX2: [[VEC_EPILOG_PH]]:
; AVX2-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
+; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
; AVX2-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX2-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX2: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
-; AVX2-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX2-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
+; AVX2-NEXT: [[WIDE_LOAD10:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
+; AVX2-NEXT: [[TMP34:%.*]] = and <4 x i8> [[WIDE_LOAD10]], splat (i8 1)
+; AVX2-NEXT: [[TMP33:%.*]] = icmp ne <4 x i8> [[TMP34]], zeroinitializer
; AVX2-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
-; AVX2-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX2-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
+; AVX2-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP33]], <4 x ptr> poison)
+; AVX2-NEXT: [[TMP37:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD11]], splat (ptr null)
+; AVX2-NEXT: [[TMP38:%.*]] = select <4 x i1> [[TMP33]], <4 x i1> [[TMP37]], <4 x i1> zeroinitializer
; AVX2-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
-; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP38]])
+; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
; AVX2-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
-; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP37:![0-9]+]]
; AVX2: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N14:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC9]]
; AVX2-NEXT: br i1 [[CMP_N14]], [[FOR_END_LOOPEXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
@@ -1636,13 +1843,13 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX512-NEXT: call void @llvm.masked.store.v64f64.p0(<64 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <64 x i1> [[TMP6]])
; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
; AVX512-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; AVX512-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP47:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP48:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX512-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX512: [[VEC_EPILOG_ITER_CHECK]]:
; AVX512-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
-; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF44]]
+; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF45]]
; AVX512: [[VEC_EPILOG_PH]]:
; AVX512-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX512-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
@@ -1662,7 +1869,7 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP34]])
; AVX512-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
; AVX512-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
-; AVX512-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP48:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP49:![0-9]+]]
; AVX512: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX512-NEXT: [[CMP_N14:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC9]]
; AVX512-NEXT: br i1 [[CMP_N14]], [[FOR_END_LOOPEXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
@@ -1941,7 +2148,7 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: [[TMP199]] = or <8 x i1> [[VEC_PHI3]], [[TMP195]]
; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 32
; AVX2-NEXT: [[TMP104:%.*]] = icmp eq i32 [[INDEX_NEXT]], 96
-; AVX2-NEXT: br i1 [[TMP104]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP38:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP104]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP39:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
; AVX2-NEXT: [[BIN_RDX:%.*]] = or <8 x i1> [[TMP197]], [[TMP196]]
; AVX2-NEXT: [[BIN_RDX4:%.*]] = or <8 x i1> [[TMP198]], [[BIN_RDX]]
@@ -1951,7 +2158,7 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: [[RDX_SELECT:%.*]] = select i1 [[TMP106]], i32 0, i32 1
; AVX2-NEXT: br i1 false, [[EXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX2: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX2-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF39:![0-9]+]]
+; AVX2-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF40:![0-9]+]]
; AVX2: [[VEC_EPILOG_PH]]:
; AVX2-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ 96, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX2-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 1, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -1990,7 +2197,7 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: [[TMP133]] = or <4 x i1> [[VEC_PHI7]], [[TMP132]]
; AVX2-NEXT: [[INDEX_NEXT8]] = add nuw i32 [[INDEX6]], 4
; AVX2-NEXT: [[TMP134:%.*]] = icmp eq i32 [[INDEX_NEXT8]], 100
-; AVX2-NEXT: br i1 [[TMP134]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP40:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP134]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP41:![0-9]+]]
; AVX2: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX2-NEXT: [[TMP135:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP133]])
; AVX2-NEXT: [[TMP136:%.*]] = freeze i1 [[TMP135]]
@@ -2041,7 +2248,7 @@ define i32 @reverse_gather(ptr %p) {
; AVX512-NEXT: [[TMP14]] = or <8 x i1> [[VEC_PHI3]], [[TMP10]]
; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 32
; AVX512-NEXT: [[TMP15:%.*]] = icmp eq i32 [[INDEX_NEXT]], 96
-; AVX512-NEXT: br i1 [[TMP15]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP50:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP15]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP51:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: [[BIN_RDX:%.*]] = or <8 x i1> [[TMP12]], [[TMP11]]
; AVX512-NEXT: [[BIN_RDX13:%.*]] = or <8 x i1> [[TMP13]], [[BIN_RDX]]
@@ -2051,7 +2258,7 @@ define i32 @reverse_gather(ptr %p) {
; AVX512-NEXT: [[RDX_SELECT:%.*]] = select i1 [[TMP17]], i32 0, i32 1
; AVX512-NEXT: br i1 false, [[EXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX512: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX512-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF51:![0-9]+]]
+; AVX512-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF52:![0-9]+]]
; AVX512: [[VEC_EPILOG_PH]]:
; AVX512-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ 96, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX512-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 1, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -2084,7 +2291,7 @@ define i32 @reverse_gather(ptr %p) {
; AVX512-NEXT: [[TMP36]] = or <4 x i1> [[VEC_PHI16]], [[TMP35]]
; AVX512-NEXT: [[INDEX_NEXT19]] = add nuw i32 [[INDEX15]], 4
; AVX512-NEXT: [[TMP37:%.*]] = icmp eq i32 [[INDEX_NEXT19]], 100
-; AVX512-NEXT: br i1 [[TMP37]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP52:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP37]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP53:![0-9]+]]
; AVX512: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX512-NEXT: [[TMP38:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP36]])
; AVX512-NEXT: [[TMP39:%.*]] = freeze i1 [[TMP38]]
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/gep-nodes-with-non-gep-inst.ll b/llvm/test/Transforms/SLPVectorizer/X86/gep-nodes-with-non-gep-inst.ll
index dfd2c4a217dd7..ab56ca7591bfe 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/gep-nodes-with-non-gep-inst.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/gep-nodes-with-non-gep-inst.ll
@@ -9,9 +9,17 @@ define void @test() {
; CHECK-NEXT: [[COND_IN_V:%.*]] = select i1 false, ptr null, ptr null
; CHECK-NEXT: br label [[BB:%.*]]
; CHECK: bb:
-; CHECK-NEXT: [[TMP0:%.*]] = call <13 x i64> @llvm.masked.load.v13i64.p0(ptr align 8 [[COND_IN_V]], <13 x i1> <i1 true, i1 false, i1 false, i1 false, i1 true, i1 false, i1 false, i1 false, i1 true, i1 false, i1 false, i1 false, i1 true>, <13 x i64> poison)
-; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <13 x i64> [[TMP0]], <13 x i64> poison, <4 x i32> <i32 0, i32 4, i32 8, i32 12>
-; CHECK-NEXT: [[TMP2:%.*]] = icmp eq <4 x i64> [[TMP1]], zeroinitializer
+; CHECK-NEXT: [[V:%.*]] = load i64, ptr [[COND_IN_V]], align 8
+; CHECK-NEXT: [[BV:%.*]] = icmp eq i64 [[V]], 0
+; CHECK-NEXT: [[IN_1:%.*]] = getelementptr i64, ptr [[COND_IN_V]], i64 4
+; CHECK-NEXT: [[V_1:%.*]] = load i64, ptr [[IN_1]], align 8
+; CHECK-NEXT: [[BV_1:%.*]] = icmp eq i64 [[V_1]], 0
+; CHECK-NEXT: [[IN_2:%.*]] = getelementptr i64, ptr [[COND_IN_V]], i64 8
+; CHECK-NEXT: [[V_2:%.*]] = load i64, ptr [[IN_2]], align 8
+; CHECK-NEXT: [[BV_2:%.*]] = icmp eq i64 [[V_2]], 0
+; CHECK-NEXT: [[IN_3:%.*]] = getelementptr i64, ptr [[COND_IN_V]], i64 12
+; CHECK-NEXT: [[V_3:%.*]] = load i64, ptr [[IN_3]], align 8
+; CHECK-NEXT: [[BV_3:%.*]] = icmp eq i64 [[V_3]], 0
; CHECK-NEXT: ret void
;
; CHECK-SLP-THRESHOLD-LABEL: define void @test
@@ -20,9 +28,15 @@ define void @test() {
; CHECK-SLP-THRESHOLD-NEXT: [[COND_IN_V:%.*]] = select i1 false, ptr null, ptr null
; CHECK-SLP-THRESHOLD-NEXT: br label [[BB:%.*]]
; CHECK-SLP-THRESHOLD: bb:
-; CHECK-SLP-THRESHOLD-NEXT: [[TMP0:%.*]] = call <13 x i64> @llvm.masked.load.v13i64.p0(ptr align 8 [[COND_IN_V]], <13 x i1> <i1 true, i1 false, i1 false, i1 false, i1 true, i1 false, i1 false, i1 false, i1 true, i1 false, i1 false, i1 false, i1 true>, <13 x i64> poison)
-; CHECK-SLP-THRESHOLD-NEXT: [[TMP1:%.*]] = shufflevector <13 x i64> [[TMP0]], <13 x i64> poison, <4 x i32> <i32 0, i32 4, i32 8, i32 12>
-; CHECK-SLP-THRESHOLD-NEXT: [[TMP2:%.*]] = icmp eq <4 x i64> [[TMP1]], zeroinitializer
+; CHECK-SLP-THRESHOLD-NEXT: [[IN_2:%.*]] = getelementptr i64, ptr [[COND_IN_V]], i64 8
+; CHECK-SLP-THRESHOLD-NEXT: [[TMP0:%.*]] = call <5 x i64> @llvm.masked.load.v5i64.p0(ptr align 8 [[COND_IN_V]], <5 x i1> <i1 true, i1 false, i1 false, i1 false, i1 true>, <5 x i64> poison)
+; CHECK-SLP-THRESHOLD-NEXT: [[TMP1:%.*]] = shufflevector <5 x i64> [[TMP0]], <5 x i64> poison, <2 x i32> <i32 4, i32 0>
+; CHECK-SLP-THRESHOLD-NEXT: [[TMP2:%.*]] = call <5 x i64> @llvm.masked.load.v5i64.p0(ptr align 8 [[IN_2]], <5 x i1> <i1 true, i1 false, i1 false, i1 false, i1 true>, <5 x i64> poison)
+; CHECK-SLP-THRESHOLD-NEXT: [[TMP3:%.*]] = shufflevector <5 x i64> [[TMP2]], <5 x i64> poison, <2 x i32> <i32 4, i32 0>
+; CHECK-SLP-THRESHOLD-NEXT: [[TMP4:%.*]] = shufflevector <2 x i64> [[TMP3]], <2 x i64> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; CHECK-SLP-THRESHOLD-NEXT: [[TMP5:%.*]] = shufflevector <2 x i64> [[TMP1]], <2 x i64> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; CHECK-SLP-THRESHOLD-NEXT: [[TMP6:%.*]] = shufflevector <5 x i64> [[TMP2]], <5 x i64> [[TMP0]], <4 x i32> <i32 4, i32 0, i32 9, i32 5>
+; CHECK-SLP-THRESHOLD-NEXT: [[TMP7:%.*]] = icmp eq <4 x i64> [[TMP6]], zeroinitializer
; CHECK-SLP-THRESHOLD-NEXT: ret void
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/pr47629-inseltpoison.ll b/llvm/test/Transforms/SLPVectorizer/X86/pr47629-inseltpoison.ll
index 02d7698fc040e..01ec3f173ce49 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/pr47629-inseltpoison.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/pr47629-inseltpoison.ll
@@ -286,11 +286,30 @@ define void @gather_load_3(ptr noalias nocapture %0, ptr noalias nocapture reado
;
; AVX2-LABEL: define void @gather_load_3(
; AVX2-SAME: ptr noalias captures(none) [[TMP0:%.*]], ptr noalias readonly captures(none) [[TMP1:%.*]]) #[[ATTR0]] {
-; AVX2-NEXT: [[TMP3:%.*]] = call <22 x i32> @llvm.masked.load.v22i32.p0(ptr align 4 [[TMP1]], <22 x i1> <i1 true, i1 false, i1 false, i1 false, i1 true, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 false, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true>, <22 x i32> poison), !tbaa [[SHORT_TBAA0]]
-; AVX2-NEXT: [[TMP4:%.*]] = shufflevector <22 x i32> [[TMP3]], <22 x i32> poison, <8 x i32> <i32 0, i32 4, i32 6, i32 9, i32 11, i32 15, i32 18, i32 21>
-; AVX2-NEXT: [[TMP5:%.*]] = add <8 x i32> [[TMP4]], <i32 1, i32 3, i32 3, i32 2, i32 2, i32 4, i32 1, i32 4>
-; AVX2-NEXT: [[TMP6:%.*]] = shufflevector <8 x i32> [[TMP5]], <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 6, i32 3, i32 2, i32 7>
-; AVX2-NEXT: store <8 x i32> [[TMP6]], ptr [[TMP0]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP3:%.*]] = load i32, ptr [[TMP1]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 11
+; AVX2-NEXT: [[TMP5:%.*]] = load i32, ptr [[TMP4]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 4
+; AVX2-NEXT: [[TMP7:%.*]] = load i32, ptr [[TMP6]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 15
+; AVX2-NEXT: [[TMP9:%.*]] = load i32, ptr [[TMP8]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 18
+; AVX2-NEXT: [[TMP11:%.*]] = load i32, ptr [[TMP10]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP12:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 6
+; AVX2-NEXT: [[TMP13:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 4 [[TMP12]], <4 x i1> <i1 true, i1 false, i1 false, i1 true>, <4 x i32> poison), !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP14:%.*]] = shufflevector <4 x i32> [[TMP13]], <4 x i32> poison, <2 x i32> <i32 0, i32 3>
+; AVX2-NEXT: [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 21
+; AVX2-NEXT: [[TMP16:%.*]] = load i32, ptr [[TMP15]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP17:%.*]] = insertelement <8 x i32> poison, i32 [[TMP3]], i64 0
+; AVX2-NEXT: [[TMP18:%.*]] = insertelement <8 x i32> [[TMP17]], i32 [[TMP5]], i64 1
+; AVX2-NEXT: [[TMP19:%.*]] = insertelement <8 x i32> [[TMP18]], i32 [[TMP7]], i64 2
+; AVX2-NEXT: [[TMP20:%.*]] = insertelement <8 x i32> [[TMP19]], i32 [[TMP9]], i64 3
+; AVX2-NEXT: [[TMP21:%.*]] = insertelement <8 x i32> [[TMP20]], i32 [[TMP11]], i64 4
+; AVX2-NEXT: [[TMP22:%.*]] = shufflevector <2 x i32> [[TMP14]], <2 x i32> poison, <8 x i32> <i32 1, i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP23:%.*]] = shufflevector <8 x i32> [[TMP21]], <8 x i32> [[TMP22]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 8, i32 9, i32 poison>
+; AVX2-NEXT: [[TMP24:%.*]] = insertelement <8 x i32> [[TMP23]], i32 [[TMP16]], i64 7
+; AVX2-NEXT: [[TMP25:%.*]] = add <8 x i32> [[TMP24]], <i32 1, i32 2, i32 3, i32 4, i32 1, i32 2, i32 3, i32 4>
+; AVX2-NEXT: store <8 x i32> [[TMP25]], ptr [[TMP0]], align 4, !tbaa [[SHORT_TBAA0]]
; AVX2-NEXT: ret void
;
; AVX512F-LABEL: define void @gather_load_3(
@@ -426,11 +445,30 @@ define void @gather_load_4(ptr noalias nocapture %t0, ptr noalias nocapture read
;
; AVX2-LABEL: define void @gather_load_4(
; AVX2-SAME: ptr noalias captures(none) [[T0:%.*]], ptr noalias readonly captures(none) [[T1:%.*]]) #[[ATTR0]] {
-; AVX2-NEXT: [[TMP1:%.*]] = call <22 x i32> @llvm.masked.load.v22i32.p0(ptr align 4 [[T1]], <22 x i1> <i1 true, i1 false, i1 false, i1 false, i1 true, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 false, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true>, <22 x i32> poison), !tbaa [[SHORT_TBAA0]]
-; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <22 x i32> [[TMP1]], <22 x i32> poison, <8 x i32> <i32 0, i32 4, i32 6, i32 9, i32 11, i32 15, i32 18, i32 21>
-; AVX2-NEXT: [[TMP3:%.*]] = add <8 x i32> [[TMP2]], <i32 1, i32 3, i32 3, i32 2, i32 2, i32 4, i32 1, i32 4>
-; AVX2-NEXT: [[TMP4:%.*]] = shufflevector <8 x i32> [[TMP3]], <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 6, i32 3, i32 2, i32 7>
-; AVX2-NEXT: store <8 x i32> [[TMP4]], ptr [[T0]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[T6:%.*]] = getelementptr inbounds i32, ptr [[T1]], i64 11
+; AVX2-NEXT: [[T10:%.*]] = getelementptr inbounds i32, ptr [[T1]], i64 4
+; AVX2-NEXT: [[T14:%.*]] = getelementptr inbounds i32, ptr [[T1]], i64 15
+; AVX2-NEXT: [[T18:%.*]] = getelementptr inbounds i32, ptr [[T1]], i64 18
+; AVX2-NEXT: [[T26:%.*]] = getelementptr inbounds i32, ptr [[T1]], i64 6
+; AVX2-NEXT: [[T30:%.*]] = getelementptr inbounds i32, ptr [[T1]], i64 21
+; AVX2-NEXT: [[T3:%.*]] = load i32, ptr [[T1]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[T7:%.*]] = load i32, ptr [[T6]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[T11:%.*]] = load i32, ptr [[T10]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[T15:%.*]] = load i32, ptr [[T14]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[T19:%.*]] = load i32, ptr [[T18]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP1:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 4 [[T26]], <4 x i1> <i1 true, i1 false, i1 false, i1 true>, <4 x i32> poison), !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <2 x i32> <i32 0, i32 3>
+; AVX2-NEXT: [[T31:%.*]] = load i32, ptr [[T30]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP3:%.*]] = insertelement <8 x i32> poison, i32 [[T3]], i64 0
+; AVX2-NEXT: [[TMP4:%.*]] = insertelement <8 x i32> [[TMP3]], i32 [[T7]], i64 1
+; AVX2-NEXT: [[TMP5:%.*]] = insertelement <8 x i32> [[TMP4]], i32 [[T11]], i64 2
+; AVX2-NEXT: [[TMP6:%.*]] = insertelement <8 x i32> [[TMP5]], i32 [[T15]], i64 3
+; AVX2-NEXT: [[TMP7:%.*]] = insertelement <8 x i32> [[TMP6]], i32 [[T19]], i64 4
+; AVX2-NEXT: [[TMP8:%.*]] = shufflevector <2 x i32> [[TMP2]], <2 x i32> poison, <8 x i32> <i32 1, i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP9:%.*]] = shufflevector <8 x i32> [[TMP7]], <8 x i32> [[TMP8]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 8, i32 9, i32 poison>
+; AVX2-NEXT: [[TMP10:%.*]] = insertelement <8 x i32> [[TMP9]], i32 [[T31]], i64 7
+; AVX2-NEXT: [[TMP11:%.*]] = add <8 x i32> [[TMP10]], <i32 1, i32 2, i32 3, i32 4, i32 1, i32 2, i32 3, i32 4>
+; AVX2-NEXT: store <8 x i32> [[TMP11]], ptr [[T0]], align 4, !tbaa [[SHORT_TBAA0]]
; AVX2-NEXT: ret void
;
; AVX512F-LABEL: define void @gather_load_4(
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/pr47629.ll b/llvm/test/Transforms/SLPVectorizer/X86/pr47629.ll
index d22e056b4bba4..2a8480baf0755 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/pr47629.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/pr47629.ll
@@ -286,11 +286,30 @@ define void @gather_load_3(ptr noalias nocapture %0, ptr noalias nocapture reado
;
; AVX2-LABEL: define void @gather_load_3(
; AVX2-SAME: ptr noalias captures(none) [[TMP0:%.*]], ptr noalias readonly captures(none) [[TMP1:%.*]]) #[[ATTR0]] {
-; AVX2-NEXT: [[TMP3:%.*]] = call <22 x i32> @llvm.masked.load.v22i32.p0(ptr align 4 [[TMP1]], <22 x i1> <i1 true, i1 false, i1 false, i1 false, i1 true, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 false, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true>, <22 x i32> poison), !tbaa [[SHORT_TBAA0]]
-; AVX2-NEXT: [[TMP4:%.*]] = shufflevector <22 x i32> [[TMP3]], <22 x i32> poison, <8 x i32> <i32 0, i32 4, i32 6, i32 9, i32 11, i32 15, i32 18, i32 21>
-; AVX2-NEXT: [[TMP5:%.*]] = add <8 x i32> [[TMP4]], <i32 1, i32 3, i32 3, i32 2, i32 2, i32 4, i32 1, i32 4>
-; AVX2-NEXT: [[TMP6:%.*]] = shufflevector <8 x i32> [[TMP5]], <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 6, i32 3, i32 2, i32 7>
-; AVX2-NEXT: store <8 x i32> [[TMP6]], ptr [[TMP0]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP3:%.*]] = load i32, ptr [[TMP1]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 11
+; AVX2-NEXT: [[TMP5:%.*]] = load i32, ptr [[TMP4]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 4
+; AVX2-NEXT: [[TMP7:%.*]] = load i32, ptr [[TMP6]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 15
+; AVX2-NEXT: [[TMP9:%.*]] = load i32, ptr [[TMP8]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 18
+; AVX2-NEXT: [[TMP11:%.*]] = load i32, ptr [[TMP10]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP12:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 6
+; AVX2-NEXT: [[TMP13:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 4 [[TMP12]], <4 x i1> <i1 true, i1 false, i1 false, i1 true>, <4 x i32> poison), !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP14:%.*]] = shufflevector <4 x i32> [[TMP13]], <4 x i32> poison, <2 x i32> <i32 0, i32 3>
+; AVX2-NEXT: [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 21
+; AVX2-NEXT: [[TMP16:%.*]] = load i32, ptr [[TMP15]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP17:%.*]] = insertelement <8 x i32> poison, i32 [[TMP3]], i64 0
+; AVX2-NEXT: [[TMP18:%.*]] = insertelement <8 x i32> [[TMP17]], i32 [[TMP5]], i64 1
+; AVX2-NEXT: [[TMP19:%.*]] = insertelement <8 x i32> [[TMP18]], i32 [[TMP7]], i64 2
+; AVX2-NEXT: [[TMP20:%.*]] = insertelement <8 x i32> [[TMP19]], i32 [[TMP9]], i64 3
+; AVX2-NEXT: [[TMP21:%.*]] = insertelement <8 x i32> [[TMP20]], i32 [[TMP11]], i64 4
+; AVX2-NEXT: [[TMP22:%.*]] = shufflevector <2 x i32> [[TMP14]], <2 x i32> poison, <8 x i32> <i32 1, i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP23:%.*]] = shufflevector <8 x i32> [[TMP21]], <8 x i32> [[TMP22]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 8, i32 9, i32 poison>
+; AVX2-NEXT: [[TMP24:%.*]] = insertelement <8 x i32> [[TMP23]], i32 [[TMP16]], i64 7
+; AVX2-NEXT: [[TMP25:%.*]] = add <8 x i32> [[TMP24]], <i32 1, i32 2, i32 3, i32 4, i32 1, i32 2, i32 3, i32 4>
+; AVX2-NEXT: store <8 x i32> [[TMP25]], ptr [[TMP0]], align 4, !tbaa [[SHORT_TBAA0]]
; AVX2-NEXT: ret void
;
; AVX512F-LABEL: define void @gather_load_3(
@@ -426,11 +445,30 @@ define void @gather_load_4(ptr noalias nocapture %t0, ptr noalias nocapture read
;
; AVX2-LABEL: define void @gather_load_4(
; AVX2-SAME: ptr noalias captures(none) [[T0:%.*]], ptr noalias readonly captures(none) [[T1:%.*]]) #[[ATTR0]] {
-; AVX2-NEXT: [[TMP1:%.*]] = call <22 x i32> @llvm.masked.load.v22i32.p0(ptr align 4 [[T1]], <22 x i1> <i1 true, i1 false, i1 false, i1 false, i1 true, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 false, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true, i1 false, i1 false, i1 true>, <22 x i32> poison), !tbaa [[SHORT_TBAA0]]
-; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <22 x i32> [[TMP1]], <22 x i32> poison, <8 x i32> <i32 0, i32 4, i32 6, i32 9, i32 11, i32 15, i32 18, i32 21>
-; AVX2-NEXT: [[TMP3:%.*]] = add <8 x i32> [[TMP2]], <i32 1, i32 3, i32 3, i32 2, i32 2, i32 4, i32 1, i32 4>
-; AVX2-NEXT: [[TMP4:%.*]] = shufflevector <8 x i32> [[TMP3]], <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 6, i32 3, i32 2, i32 7>
-; AVX2-NEXT: store <8 x i32> [[TMP4]], ptr [[T0]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[T6:%.*]] = getelementptr inbounds i32, ptr [[T1]], i64 11
+; AVX2-NEXT: [[T10:%.*]] = getelementptr inbounds i32, ptr [[T1]], i64 4
+; AVX2-NEXT: [[T14:%.*]] = getelementptr inbounds i32, ptr [[T1]], i64 15
+; AVX2-NEXT: [[T18:%.*]] = getelementptr inbounds i32, ptr [[T1]], i64 18
+; AVX2-NEXT: [[T26:%.*]] = getelementptr inbounds i32, ptr [[T1]], i64 6
+; AVX2-NEXT: [[T30:%.*]] = getelementptr inbounds i32, ptr [[T1]], i64 21
+; AVX2-NEXT: [[T3:%.*]] = load i32, ptr [[T1]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[T7:%.*]] = load i32, ptr [[T6]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[T11:%.*]] = load i32, ptr [[T10]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[T15:%.*]] = load i32, ptr [[T14]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[T19:%.*]] = load i32, ptr [[T18]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP1:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 4 [[T26]], <4 x i1> <i1 true, i1 false, i1 false, i1 true>, <4 x i32> poison), !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <2 x i32> <i32 0, i32 3>
+; AVX2-NEXT: [[T31:%.*]] = load i32, ptr [[T30]], align 4, !tbaa [[SHORT_TBAA0]]
+; AVX2-NEXT: [[TMP3:%.*]] = insertelement <8 x i32> poison, i32 [[T3]], i64 0
+; AVX2-NEXT: [[TMP4:%.*]] = insertelement <8 x i32> [[TMP3]], i32 [[T7]], i64 1
+; AVX2-NEXT: [[TMP5:%.*]] = insertelement <8 x i32> [[TMP4]], i32 [[T11]], i64 2
+; AVX2-NEXT: [[TMP6:%.*]] = insertelement <8 x i32> [[TMP5]], i32 [[T15]], i64 3
+; AVX2-NEXT: [[TMP7:%.*]] = insertelement <8 x i32> [[TMP6]], i32 [[T19]], i64 4
+; AVX2-NEXT: [[TMP8:%.*]] = shufflevector <2 x i32> [[TMP2]], <2 x i32> poison, <8 x i32> <i32 1, i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[TMP9:%.*]] = shufflevector <8 x i32> [[TMP7]], <8 x i32> [[TMP8]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 8, i32 9, i32 poison>
+; AVX2-NEXT: [[TMP10:%.*]] = insertelement <8 x i32> [[TMP9]], i32 [[T31]], i64 7
+; AVX2-NEXT: [[TMP11:%.*]] = add <8 x i32> [[TMP10]], <i32 1, i32 2, i32 3, i32 4, i32 1, i32 2, i32 3, i32 4>
+; AVX2-NEXT: store <8 x i32> [[TMP11]], ptr [[T0]], align 4, !tbaa [[SHORT_TBAA0]]
; AVX2-NEXT: ret void
;
; AVX512F-LABEL: define void @gather_load_4(
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/remark-masked-loads-consecutive-loads-same-ptr.ll b/llvm/test/Transforms/SLPVectorizer/X86/remark-masked-loads-consecutive-loads-same-ptr.ll
index 23a901fd768b3..0a34004b77fa3 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/remark-masked-loads-consecutive-loads-same-ptr.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/remark-masked-loads-consecutive-loads-same-ptr.ll
@@ -15,9 +15,16 @@
define void @test(ptr noalias %p, ptr noalias %p1) {
; CHECK-LABEL: @test(
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[TMP0:%.*]] = call <35 x i32> @llvm.masked.load.v35i32.p0(ptr align 4 [[P:%.*]], <35 x i1> <i1 true, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 true, i1 true, i1 true>, <35 x i32> poison)
-; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <35 x i32> [[TMP0]], <35 x i32> poison, <4 x i32> <i32 0, i32 32, i32 33, i32 34>
+; CHECK-NEXT: [[I:%.*]] = load i32, ptr [[P:%.*]], align 4
+; CHECK-NEXT: [[ARRAYIDX4:%.*]] = getelementptr i32, ptr [[P]], i64 32
+; CHECK-NEXT: [[I2:%.*]] = load i32, ptr [[ARRAYIDX4]], align 4
+; CHECK-NEXT: [[ARRAYIDX11:%.*]] = getelementptr i32, ptr [[P]], i64 33
+; CHECK-NEXT: [[TMP0:%.*]] = load <2 x i32>, ptr [[ARRAYIDX11]], align 4
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr [[P]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <4 x i32> poison, i32 [[I]], i64 0
+; CHECK-NEXT: [[TMP3:%.*]] = insertelement <4 x i32> [[TMP2]], i32 [[I2]], i64 1
+; CHECK-NEXT: [[TMP6:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <4 x i32> [[TMP3]], <4 x i32> [[TMP6]], <4 x i32> <i32 0, i32 1, i32 4, i32 5>
; CHECK-NEXT: [[TMP5:%.*]] = add nsw <4 x i32> [[TMP4]], [[TMP1]]
; CHECK-NEXT: store <4 x i32> [[TMP5]], ptr [[P1:%.*]], align 4
; CHECK-NEXT: ret void
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/scatter-vectorize-reused-pointer.ll b/llvm/test/Transforms/SLPVectorizer/X86/scatter-vectorize-reused-pointer.ll
index 77084f5b97e7d..39acbabf94ac0 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/scatter-vectorize-reused-pointer.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/scatter-vectorize-reused-pointer.ll
@@ -5,12 +5,16 @@ define void @test(i1 %c, ptr %arg) {
; CHECK-LABEL: @test(
; CHECK-NEXT: br i1 [[C:%.*]], label [[IF:%.*]], label [[ELSE:%.*]]
; CHECK: if:
-; CHECK-NEXT: [[TMP1:%.*]] = call <5 x i64> @llvm.masked.load.v5i64.p0(ptr align 8 [[ARG:%.*]], <5 x i1> <i1 true, i1 true, i1 false, i1 true, i1 true>, <5 x i64> poison)
-; CHECK-NEXT: [[TMP6:%.*]] = shufflevector <5 x i64> [[TMP1]], <5 x i64> poison, <4 x i32> <i32 0, i32 1, i32 3, i32 4>
+; CHECK-NEXT: [[ARG2_2:%.*]] = getelementptr inbounds i8, ptr [[ARG:%.*]], i64 24
+; CHECK-NEXT: [[TMP1:%.*]] = load <2 x i64>, ptr [[ARG]], align 8
+; CHECK-NEXT: [[TMP2:%.*]] = load <2 x i64>, ptr [[ARG2_2]], align 8
+; CHECK-NEXT: [[TMP6:%.*]] = shufflevector <2 x i64> [[TMP2]], <2 x i64> [[TMP1]], <4 x i32> <i32 1, i32 0, i32 3, i32 2>
; CHECK-NEXT: br label [[JOIN:%.*]]
; CHECK: else:
-; CHECK-NEXT: [[TMP3:%.*]] = call <5 x i64> @llvm.masked.load.v5i64.p0(ptr align 8 [[ARG]], <5 x i1> <i1 true, i1 true, i1 false, i1 true, i1 true>, <5 x i64> poison)
-; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <5 x i64> [[TMP3]], <5 x i64> poison, <4 x i32> <i32 0, i32 1, i32 3, i32 4>
+; CHECK-NEXT: [[ARG_2:%.*]] = getelementptr inbounds i8, ptr [[ARG]], i64 24
+; CHECK-NEXT: [[TMP4:%.*]] = load <2 x i64>, ptr [[ARG]], align 8
+; CHECK-NEXT: [[TMP5:%.*]] = load <2 x i64>, ptr [[ARG_2]], align 8
+; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <2 x i64> [[TMP5]], <2 x i64> [[TMP4]], <4 x i32> <i32 1, i32 0, i32 3, i32 2>
; CHECK-NEXT: br label [[JOIN]]
; CHECK: join:
; CHECK-NEXT: [[TMP13:%.*]] = phi <4 x i64> [ [[TMP6]], [[IF]] ], [ [[TMP12]], [[ELSE]] ]
More information about the llvm-commits
mailing list