[llvm] [X86] Enable MaximizeBandwidth by default and price predicate mask-expansion fanout (PR #201666)
Sumukh J Bharadwaj via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 9 12:53:34 PDT 2026
https://github.com/amd-subharad updated https://github.com/llvm/llvm-project/pull/201666
>From 3265a6f3df979de78bc23b30d0c856ef14549a56 Mon Sep 17 00:00:00 2001
From: Sumukh Bharadwaj <Sumukh.Bharadwaj at amd.com>
Date: Tue, 18 Aug 2026 03:35:20 +0530
Subject: [PATCH 1/2] [X86][LoopVectorize] Enable MaximizeBandwidth by default
on X86
Enable the MaximizeBandwidth heuristic by default for X86 fixed-width
vectors via shouldMaximizeVectorBandwidth(), matching the AArch64 Neon
precedent, so the loop vectorizer can select a VF from the smallest
element type in mixed-width loops.
Existing LoopVectorize/X86 and PhaseOrdering/X86 tests whose VF changes
under the new default have their CHECK lines regenerated with MaxBW
enabled, so they validate the new behavior directly rather than being
opted out with -vectorizer-maximize-bandwidth=false. A focused
maxbw-cast-cost.ll pins the VF-from-smallest-type selection.
Enabling the heuristic is neutral-to-positive in benchmarking. The one
mispricing it exposes -- the X86 per-part cost tables under-counting
predicate mask-expansion fanout, which let masked / gather-scatter loops
over-widen -- is addressed by the companion X86 TTI cost-model commit in
this series, so the fix stays in the cost model rather than a generic
LV-side floor or vectorizer-level workaround.
---
llvm/lib/Target/X86/X86TargetTransformInfo.h | 6 +
.../X86/conditional-scalar-assignment.ll | 109 ++-
.../X86/cost-conditional-branches.ll | 375 +-------
.../LoopVectorize/X86/cost-model.ll | 26 +-
.../X86/epilog-vectorization-inductions.ll | 45 +-
.../Transforms/LoopVectorize/X86/funclet.ll | 15 +-
.../LoopVectorize/X86/gcc-examples.ll | 144 ++-
.../LoopVectorize/X86/induction-costs.ll | 60 +-
.../LoopVectorize/X86/masked_load_store.ll | 883 +++++++-----------
.../LoopVectorize/X86/maxbw-cast-cost.ll | 353 +++++++
.../Transforms/LoopVectorize/X86/no_fpmath.ll | 2 +-
.../X86/no_fpmath_with_hotness.ll | 2 +-
.../X86/nondetermisitic-widening-cost.ll | 48 +-
.../X86/pr131359-dead-for-splice.ll | 46 +-
.../Transforms/LoopVectorize/X86/pr47437.ll | 85 +-
.../LoopVectorize/X86/reduction-crash.ll | 24 +-
.../X86/replicating-load-store-costs.ll | 303 +++---
.../LoopVectorize/X86/strided_load_cost.ll | 124 ++-
.../X86/vector_ptr_load_store.ll | 4 +-
.../X86/vectorization-remarks-loopid-dbg.ll | 149 ++-
.../X86/vectorization-remarks.ll | 171 +++-
.../PhaseOrdering/X86/pixel-splat.ll | 55 +-
.../X86/preserve-access-group.ll | 28 +-
.../X86/vector-reduction-known-first-value.ll | 274 +++++-
24 files changed, 1979 insertions(+), 1352 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/X86/maxbw-cast-cost.ll
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.h b/llvm/lib/Target/X86/X86TargetTransformInfo.h
index 22171f5469d98..4d769f0f49c9f 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.h
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.h
@@ -64,6 +64,12 @@ class X86TTIImpl final : public BasicTTIImplBase<X86TTIImpl> {
TypeSize
getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override;
unsigned getLoadStoreVecRegBitWidth(unsigned AS) const override;
+ bool shouldMaximizeVectorBandwidth(
+ TargetTransformInfo::RegisterKind K) const override {
+ assert(K != TargetTransformInfo::RGK_Scalar &&
+ "Expected vector register kind");
+ return K == TargetTransformInfo::RGK_FixedWidthVector;
+ }
unsigned getMaxInterleaveFactor(ElementCount VF,
bool HasUnorderedReductions) const override;
InstructionCost getArithmeticInstrCost(
diff --git a/llvm/test/Transforms/LoopVectorize/X86/conditional-scalar-assignment.ll b/llvm/test/Transforms/LoopVectorize/X86/conditional-scalar-assignment.ll
index f6bd69acb775b..ec3bfaecf4941 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/conditional-scalar-assignment.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/conditional-scalar-assignment.ll
@@ -388,22 +388,59 @@ define i32 @multi_use_cmp_for_csa_int_select(i64 %N, ptr %data, i32 %a) {
; X86-LABEL: define i32 @multi_use_cmp_for_csa_int_select(
; X86-SAME: i64 [[N:%.*]], ptr [[DATA:%.*]], i32 [[A:%.*]]) {
; X86-NEXT: [[ENTRY:.*]]:
+; X86-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; X86-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; X86: [[VECTOR_PH]]:
+; X86-NEXT: [[TMP0:%.*]] = and i64 [[N]], 3
+; X86-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; X86-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[A]], i64 0
+; X86-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
; X86-NEXT: br label %[[LOOP:.*]]
; X86: [[LOOP]]:
-; X86-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; X86-NEXT: [[DATA_PHI:%.*]] = phi i32 [ -1, %[[ENTRY]] ], [ [[SELECT_DATA:%.*]], %[[LOOP]] ]
-; X86-NEXT: [[IDX_PHI:%.*]] = phi i64 [ -1, %[[ENTRY]] ], [ [[SELECT_IDX:%.*]], %[[LOOP]] ]
+; X86-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP]] ]
+; X86-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[LOOP]] ]
+; X86-NEXT: [[VEC_PHI:%.*]] = phi <4 x i32> [ splat (i32 -1), %[[VECTOR_PH]] ], [ [[TMP7:%.*]], %[[LOOP]] ]
+; X86-NEXT: [[TMP1:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP6:%.*]], %[[LOOP]] ]
+; X86-NEXT: [[VEC_PHI1:%.*]] = phi <4 x i64> [ splat (i64 -9223372036854775808), %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[LOOP]] ]
; X86-NEXT: [[LD_ADDR:%.*]] = getelementptr inbounds i32, ptr [[DATA]], i64 [[IV]]
-; X86-NEXT: [[LD:%.*]] = load i32, ptr [[LD_ADDR]], align 4
+; X86-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[LD_ADDR]], align 4
+; X86-NEXT: [[TMP3:%.*]] = icmp slt <4 x i32> [[BROADCAST_SPLAT]], [[WIDE_LOAD]]
+; X86-NEXT: [[TMP4:%.*]] = freeze <4 x i1> [[TMP3]]
+; X86-NEXT: [[TMP5:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP4]])
+; X86-NEXT: [[TMP6]] = select i1 [[TMP5]], <4 x i1> [[TMP3]], <4 x i1> [[TMP1]]
+; X86-NEXT: [[TMP7]] = select i1 [[TMP5]], <4 x i32> [[WIDE_LOAD]], <4 x i32> [[VEC_PHI]]
+; X86-NEXT: [[TMP8]] = select <4 x i1> [[TMP3]], <4 x i64> [[VEC_IND]], <4 x i64> [[VEC_PHI1]]
+; X86-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
+; X86-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; X86-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; X86-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP6:![0-9]+]]
+; X86: [[MIDDLE_BLOCK]]:
+; X86-NEXT: [[TMP10:%.*]] = call i32 @llvm.experimental.vector.extract.last.active.v4i32(<4 x i32> [[TMP7]], <4 x i1> [[TMP6]], i32 -1)
+; X86-NEXT: [[TMP11:%.*]] = call i64 @llvm.vector.reduce.smax.v4i64(<4 x i64> [[TMP8]])
+; X86-NEXT: [[TMP12:%.*]] = icmp ne i64 [[TMP11]], -9223372036854775808
+; X86-NEXT: [[TMP13:%.*]] = select i1 [[TMP12]], i64 [[TMP11]], i64 -1
+; X86-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; X86-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; X86: [[SCALAR_PH]]:
+; X86-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; X86-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP10]], %[[MIDDLE_BLOCK]] ], [ -1, %[[ENTRY]] ]
+; X86-NEXT: [[BC_MERGE_RDX2:%.*]] = phi i64 [ [[TMP13]], %[[MIDDLE_BLOCK]] ], [ -1, %[[ENTRY]] ]
+; X86-NEXT: br label %[[LOOP1:.*]]
+; X86: [[LOOP1]]:
+; X86-NEXT: [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; X86-NEXT: [[DATA_PHI:%.*]] = phi i32 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[SELECT_DATA:%.*]], %[[LOOP1]] ]
+; X86-NEXT: [[IDX_PHI:%.*]] = phi i64 [ [[BC_MERGE_RDX2]], %[[SCALAR_PH]] ], [ [[SELECT_IDX:%.*]], %[[LOOP1]] ]
+; X86-NEXT: [[LD_ADDR1:%.*]] = getelementptr inbounds i32, ptr [[DATA]], i64 [[IV1]]
+; X86-NEXT: [[LD:%.*]] = load i32, ptr [[LD_ADDR1]], align 4
; X86-NEXT: [[SELECT_CMP:%.*]] = icmp slt i32 [[A]], [[LD]]
; X86-NEXT: [[SELECT_DATA]] = select i1 [[SELECT_CMP]], i32 [[LD]], i32 [[DATA_PHI]]
-; X86-NEXT: [[SELECT_IDX]] = select i1 [[SELECT_CMP]], i64 [[IV]], i64 [[IDX_PHI]]
-; X86-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; X86-NEXT: [[SELECT_IDX]] = select i1 [[SELECT_CMP]], i64 [[IV1]], i64 [[IDX_PHI]]
+; X86-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; X86-NEXT: [[EXIT_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT:.*]], label %[[LOOP]]
+; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP7:![0-9]+]]
; X86: [[EXIT]]:
-; X86-NEXT: [[SELECT_DATA_LCSSA:%.*]] = phi i32 [ [[SELECT_DATA]], %[[LOOP]] ]
-; X86-NEXT: [[SELECT_IDX_LCSSA:%.*]] = phi i64 [ [[SELECT_IDX]], %[[LOOP]] ]
+; X86-NEXT: [[SELECT_DATA_LCSSA:%.*]] = phi i32 [ [[SELECT_DATA]], %[[LOOP1]] ], [ [[TMP10]], %[[MIDDLE_BLOCK]] ]
+; X86-NEXT: [[SELECT_IDX_LCSSA:%.*]] = phi i64 [ [[SELECT_IDX]], %[[LOOP1]] ], [ [[TMP13]], %[[MIDDLE_BLOCK]] ]
; X86-NEXT: [[IDX:%.*]] = trunc i64 [[SELECT_IDX_LCSSA]] to i32
; X86-NEXT: [[RES:%.*]] = add i32 [[IDX]], [[SELECT_DATA_LCSSA]]
; X86-NEXT: ret i32 [[RES]]
@@ -411,42 +448,42 @@ define i32 @multi_use_cmp_for_csa_int_select(i64 %N, ptr %data, i32 %a) {
; AVX512-LABEL: define i32 @multi_use_cmp_for_csa_int_select(
; AVX512-SAME: i64 [[N:%.*]], ptr [[DATA:%.*]], i32 [[A:%.*]]) #[[ATTR0]] {
; AVX512-NEXT: [[ENTRY:.*]]:
-; AVX512-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 16
+; AVX512-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 32
; AVX512-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX512: [[VECTOR_PH]]:
-; AVX512-NEXT: [[N_MOD_VF:%.*]] = and i64 [[N]], 7
+; AVX512-NEXT: [[N_MOD_VF:%.*]] = and i64 [[N]], 15
; AVX512-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
-; AVX512-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i32> poison, i32 [[A]], i64 0
-; AVX512-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT]], <8 x i32> poison, <8 x i32> zeroinitializer
+; AVX512-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <16 x i32> poison, i32 [[A]], i64 0
+; AVX512-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <16 x i32> [[BROADCAST_SPLATINSERT]], <16 x i32> poison, <16 x i32> zeroinitializer
; AVX512-NEXT: br label %[[LOOP:.*]]
; AVX512: [[LOOP]]:
; AVX512-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[LOOP]] ]
-; AVX512-NEXT: [[VEC_IND:%.*]] = phi <8 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[LOOP]] ]
-; AVX512-NEXT: [[VEC_PHI:%.*]] = phi <8 x i32> [ splat (i32 -1), %[[VECTOR_PH]] ], [ [[TMP6:%.*]], %[[LOOP]] ]
-; AVX512-NEXT: [[TMP0:%.*]] = phi <8 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[LOOP]] ]
-; AVX512-NEXT: [[VEC_PHI1:%.*]] = phi <8 x i64> [ splat (i64 -9223372036854775808), %[[VECTOR_PH]] ], [ [[TMP7:%.*]], %[[LOOP]] ]
+; AVX512-NEXT: [[VEC_IND:%.*]] = phi <16 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9, i64 10, i64 11, i64 12, i64 13, i64 14, i64 15>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[LOOP]] ]
+; AVX512-NEXT: [[VEC_PHI:%.*]] = phi <16 x i32> [ splat (i32 -1), %[[VECTOR_PH]] ], [ [[TMP7:%.*]], %[[LOOP]] ]
+; AVX512-NEXT: [[TMP1:%.*]] = phi <16 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP6:%.*]], %[[LOOP]] ]
+; AVX512-NEXT: [[VEC_PHI1:%.*]] = phi <16 x i64> [ splat (i64 -9223372036854775808), %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[LOOP]] ]
; AVX512-NEXT: [[LD_ADDR:%.*]] = getelementptr inbounds i32, ptr [[DATA]], i64 [[IV]]
-; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[LD_ADDR]], align 4
-; AVX512-NEXT: [[TMP2:%.*]] = icmp slt <8 x i32> [[BROADCAST_SPLAT]], [[WIDE_LOAD]]
-; AVX512-NEXT: [[TMP3:%.*]] = freeze <8 x i1> [[TMP2]]
-; AVX512-NEXT: [[TMP4:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP3]])
-; AVX512-NEXT: [[TMP5]] = select i1 [[TMP4]], <8 x i1> [[TMP2]], <8 x i1> [[TMP0]]
-; AVX512-NEXT: [[TMP6]] = select i1 [[TMP4]], <8 x i32> [[WIDE_LOAD]], <8 x i32> [[VEC_PHI]]
-; AVX512-NEXT: [[TMP7]] = select <8 x i1> [[TMP2]], <8 x i64> [[VEC_IND]], <8 x i64> [[VEC_PHI1]]
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 8
-; AVX512-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <8 x i64> [[VEC_IND]], splat (i64 8)
+; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[LD_ADDR]], align 4
+; AVX512-NEXT: [[TMP3:%.*]] = icmp slt <16 x i32> [[BROADCAST_SPLAT]], [[WIDE_LOAD]]
+; AVX512-NEXT: [[TMP4:%.*]] = freeze <16 x i1> [[TMP3]]
+; AVX512-NEXT: [[TMP5:%.*]] = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> [[TMP4]])
+; AVX512-NEXT: [[TMP6]] = select i1 [[TMP5]], <16 x i1> [[TMP3]], <16 x i1> [[TMP1]]
+; AVX512-NEXT: [[TMP7]] = select i1 [[TMP5]], <16 x i32> [[WIDE_LOAD]], <16 x i32> [[VEC_PHI]]
+; AVX512-NEXT: [[TMP9]] = select <16 x i1> [[TMP3]], <16 x i64> [[VEC_IND]], <16 x i64> [[VEC_PHI1]]
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 16
+; AVX512-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <16 x i64> [[VEC_IND]], splat (i64 16)
; AVX512-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX512-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
-; AVX512-NEXT: [[TMP9:%.*]] = call i32 @llvm.experimental.vector.extract.last.active.v8i32(<8 x i32> [[TMP6]], <8 x i1> [[TMP5]], i32 -1)
-; AVX512-NEXT: [[TMP10:%.*]] = call i64 @llvm.vector.reduce.smax.v8i64(<8 x i64> [[TMP7]])
+; AVX512-NEXT: [[TMP13:%.*]] = call i32 @llvm.experimental.vector.extract.last.active.v16i32(<16 x i32> [[TMP7]], <16 x i1> [[TMP6]], i32 -1)
+; AVX512-NEXT: [[TMP10:%.*]] = call i64 @llvm.vector.reduce.smax.v16i64(<16 x i64> [[TMP9]])
; AVX512-NEXT: [[TMP11:%.*]] = icmp ne i64 [[TMP10]], -9223372036854775808
; AVX512-NEXT: [[TMP12:%.*]] = select i1 [[TMP11]], i64 [[TMP10]], i64 -1
; AVX512-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
; AVX512-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
; AVX512: [[SCALAR_PH]]:
; AVX512-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
-; AVX512-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP9]], %[[MIDDLE_BLOCK]] ], [ -1, %[[ENTRY]] ]
+; AVX512-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP13]], %[[MIDDLE_BLOCK]] ], [ -1, %[[ENTRY]] ]
; AVX512-NEXT: [[BC_MERGE_RDX2:%.*]] = phi i64 [ [[TMP12]], %[[MIDDLE_BLOCK]] ], [ -1, %[[ENTRY]] ]
; AVX512-NEXT: br label %[[LOOP1:.*]]
; AVX512: [[LOOP1]]:
@@ -462,7 +499,7 @@ define i32 @multi_use_cmp_for_csa_int_select(i64 %N, ptr %data, i32 %a) {
; AVX512-NEXT: [[EXIT_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
; AVX512-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP9:![0-9]+]]
; AVX512: [[EXIT]]:
-; AVX512-NEXT: [[SELECT_DATA_LCSSA:%.*]] = phi i32 [ [[SELECT_DATA]], %[[LOOP1]] ], [ [[TMP9]], %[[MIDDLE_BLOCK]] ]
+; AVX512-NEXT: [[SELECT_DATA_LCSSA:%.*]] = phi i32 [ [[SELECT_DATA]], %[[LOOP1]] ], [ [[TMP13]], %[[MIDDLE_BLOCK]] ]
; AVX512-NEXT: [[SELECT_IDX_LCSSA:%.*]] = phi i64 [ [[SELECT_IDX]], %[[LOOP1]] ], [ [[TMP12]], %[[MIDDLE_BLOCK]] ]
; AVX512-NEXT: [[IDX:%.*]] = trunc i64 [[SELECT_IDX_LCSSA]] to i32
; AVX512-NEXT: [[RES:%.*]] = add i32 [[IDX]], [[SELECT_DATA_LCSSA]]
@@ -652,7 +689,7 @@ define i32 @int_select_with_extra_arith_payload(i64 %N, ptr readonly %A, ptr rea
; X86-NEXT: [[TMP11]] = select i1 [[TMP9]], <4 x i32> [[WIDE_LOAD]], <4 x i32> [[VEC_PHI]]
; X86-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
; X86-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; X86-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP6:![0-9]+]]
+; X86-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
; X86: [[MIDDLE_BLOCK]]:
; X86-NEXT: [[TMP13:%.*]] = call i32 @llvm.experimental.vector.extract.last.active.v4i32(<4 x i32> [[TMP11]], <4 x i1> [[TMP10]], i32 -1)
; X86-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
@@ -677,7 +714,7 @@ define i32 @int_select_with_extra_arith_payload(i64 %N, ptr readonly %A, ptr rea
; X86-NEXT: [[SELECT_A]] = select i1 [[SELECT_CMP]], i32 [[LD_A]], i32 [[A_PHI]]
; X86-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; X86-NEXT: [[EXIT_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP7:![0-9]+]]
+; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP9:![0-9]+]]
; X86: [[EXIT]]:
; X86-NEXT: [[SELECT_A_LCSSA:%.*]] = phi i32 [ [[SELECT_A]], %[[LOOP1]] ], [ [[TMP13]], %[[MIDDLE_BLOCK]] ]
; X86-NEXT: ret i32 [[SELECT_A_LCSSA]]
@@ -793,7 +830,7 @@ define i8 @simple_csa_byte_select(i64 %N, ptr %data, i8 %a) {
; X86-NEXT: [[TMP6]] = select i1 [[TMP4]], <16 x i8> [[WIDE_LOAD]], <16 x i8> [[VEC_PHI]]
; X86-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 16
; X86-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; X86-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
+; X86-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP10:![0-9]+]]
; X86: [[MIDDLE_BLOCK]]:
; X86-NEXT: [[TMP8:%.*]] = call i8 @llvm.experimental.vector.extract.last.active.v16i8(<16 x i8> [[TMP6]], <16 x i1> [[TMP5]], i8 -1)
; X86-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
@@ -811,7 +848,7 @@ define i8 @simple_csa_byte_select(i64 %N, ptr %data, i8 %a) {
; X86-NEXT: [[SELECT_DATA]] = select i1 [[SELECT_CMP]], i8 [[LD]], i8 [[DATA_PHI]]
; X86-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; X86-NEXT: [[EXIT_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP9:![0-9]+]]
+; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP11:![0-9]+]]
; X86: [[EXIT]]:
; X86-NEXT: [[SELECT_DATA_LCSSA:%.*]] = phi i8 [ [[SELECT_DATA]], %[[LOOP1]] ], [ [[TMP8]], %[[MIDDLE_BLOCK]] ]
; X86-NEXT: ret i8 [[SELECT_DATA_LCSSA]]
@@ -915,7 +952,7 @@ define i32 @simple_csa_int_select_use_interleave(i64 %N, ptr %data, i32 %a) {
; X86-NEXT: [[TMP13]] = select i1 [[TMP4]], <4 x i32> [[WIDE_LOAD2]], <4 x i32> [[VEC_PHI1]]
; X86-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 8
; X86-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; X86-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP10:![0-9]+]]
+; X86-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP12:![0-9]+]]
; X86: [[MIDDLE_BLOCK]]:
; X86-NEXT: [[TMP8:%.*]] = call i32 @llvm.experimental.vector.extract.last.active.v4i32(<4 x i32> [[TMP6]], <4 x i1> [[TMP5]], i32 -1)
; X86-NEXT: [[TMP16:%.*]] = call i32 @llvm.experimental.vector.extract.last.active.v4i32(<4 x i32> [[TMP13]], <4 x i1> [[TMP11]], i32 [[TMP8]])
@@ -934,7 +971,7 @@ define i32 @simple_csa_int_select_use_interleave(i64 %N, ptr %data, i32 %a) {
; X86-NEXT: [[SELECT_DATA]] = select i1 [[SELECT_CMP]], i32 [[LD]], i32 [[DATA_PHI]]
; X86-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; X86-NEXT: [[EXIT_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP11:![0-9]+]]
+; X86-NEXT: br i1 [[EXIT_CMP]], label %[[EXIT]], label %[[LOOP1]], !llvm.loop [[LOOP13:![0-9]+]]
; X86: [[EXIT]]:
; X86-NEXT: [[SELECT_DATA_LCSSA:%.*]] = phi i32 [ [[SELECT_DATA]], %[[LOOP1]] ], [ [[TMP16]], %[[MIDDLE_BLOCK]] ]
; X86-NEXT: ret i32 [[SELECT_DATA_LCSSA]]
diff --git a/llvm/test/Transforms/LoopVectorize/X86/cost-conditional-branches.ll b/llvm/test/Transforms/LoopVectorize/X86/cost-conditional-branches.ll
index f0c94e0a78629..78bb964a3bb50 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/cost-conditional-branches.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/cost-conditional-branches.ll
@@ -868,386 +868,23 @@ exit:
; Test case for https://github.com/llvm/llvm-project/issues/158660.
define i64 @test_predicated_udiv(i32 %d, i1 %c) #2 {
; CHECK-LABEL: @test_predicated_udiv(
-; CHECK-NEXT: iter.check:
-; CHECK-NEXT: br i1 false, label [[VEC_EPILOG_SCALAR_PH:%.*]], label [[VECTOR_MAIN_LOOP_ITER_CHECK:%.*]]
-; CHECK: vector.main.loop.iter.check:
-; CHECK-NEXT: br i1 false, label [[VEC_EPILOG_PH:%.*]], label [[VECTOR_PH:%.*]]
-; CHECK: vector.ph:
-; CHECK-NEXT: [[TMP0:%.*]] = xor i1 [[C:%.*]], true
-; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
-; CHECK: vector.body:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[PRED_UDIV_CONTINUE62:%.*]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <32 x i32> [ <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>, [[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], [[PRED_UDIV_CONTINUE62]] ]
-; CHECK-NEXT: [[TMP1:%.*]] = call <32 x i32> @llvm.usub.sat.v32i32(<32 x i32> [[VEC_IND]], <32 x i32> splat (i32 1))
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF:%.*]], label [[PRED_UDIV_CONTINUE:%.*]]
-; CHECK: pred.udiv.if:
-; CHECK-NEXT: [[TMP2:%.*]] = extractelement <32 x i32> [[TMP1]], i64 0
-; CHECK-NEXT: [[TMP3:%.*]] = udiv i32 [[TMP2]], [[D:%.*]]
-; CHECK-NEXT: [[TMP4:%.*]] = insertelement <32 x i32> poison, i32 [[TMP3]], i64 0
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE]]
-; CHECK: pred.udiv.continue:
-; CHECK-NEXT: [[TMP5:%.*]] = phi <32 x i32> [ poison, [[VECTOR_BODY]] ], [ [[TMP4]], [[PRED_UDIV_IF]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF1:%.*]], label [[PRED_UDIV_CONTINUE2:%.*]]
-; CHECK: pred.udiv.if1:
-; CHECK-NEXT: [[TMP6:%.*]] = extractelement <32 x i32> [[TMP1]], i64 1
-; CHECK-NEXT: [[TMP7:%.*]] = udiv i32 [[TMP6]], [[D]]
-; CHECK-NEXT: [[TMP8:%.*]] = insertelement <32 x i32> [[TMP5]], i32 [[TMP7]], i64 1
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE2]]
-; CHECK: pred.udiv.continue2:
-; CHECK-NEXT: [[TMP9:%.*]] = phi <32 x i32> [ [[TMP5]], [[PRED_UDIV_CONTINUE]] ], [ [[TMP8]], [[PRED_UDIV_IF1]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF3:%.*]], label [[PRED_UDIV_CONTINUE4:%.*]]
-; CHECK: pred.udiv.if3:
-; CHECK-NEXT: [[TMP10:%.*]] = extractelement <32 x i32> [[TMP1]], i64 2
-; CHECK-NEXT: [[TMP11:%.*]] = udiv i32 [[TMP10]], [[D]]
-; CHECK-NEXT: [[TMP12:%.*]] = insertelement <32 x i32> [[TMP9]], i32 [[TMP11]], i64 2
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE4]]
-; CHECK: pred.udiv.continue4:
-; CHECK-NEXT: [[TMP13:%.*]] = phi <32 x i32> [ [[TMP9]], [[PRED_UDIV_CONTINUE2]] ], [ [[TMP12]], [[PRED_UDIV_IF3]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF5:%.*]], label [[PRED_UDIV_CONTINUE6:%.*]]
-; CHECK: pred.udiv.if5:
-; CHECK-NEXT: [[TMP14:%.*]] = extractelement <32 x i32> [[TMP1]], i64 3
-; CHECK-NEXT: [[TMP15:%.*]] = udiv i32 [[TMP14]], [[D]]
-; CHECK-NEXT: [[TMP16:%.*]] = insertelement <32 x i32> [[TMP13]], i32 [[TMP15]], i64 3
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE6]]
-; CHECK: pred.udiv.continue6:
-; CHECK-NEXT: [[TMP17:%.*]] = phi <32 x i32> [ [[TMP13]], [[PRED_UDIV_CONTINUE4]] ], [ [[TMP16]], [[PRED_UDIV_IF5]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF7:%.*]], label [[PRED_UDIV_CONTINUE8:%.*]]
-; CHECK: pred.udiv.if7:
-; CHECK-NEXT: [[TMP18:%.*]] = extractelement <32 x i32> [[TMP1]], i64 4
-; CHECK-NEXT: [[TMP19:%.*]] = udiv i32 [[TMP18]], [[D]]
-; CHECK-NEXT: [[TMP20:%.*]] = insertelement <32 x i32> [[TMP17]], i32 [[TMP19]], i64 4
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE8]]
-; CHECK: pred.udiv.continue8:
-; CHECK-NEXT: [[TMP21:%.*]] = phi <32 x i32> [ [[TMP17]], [[PRED_UDIV_CONTINUE6]] ], [ [[TMP20]], [[PRED_UDIV_IF7]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF9:%.*]], label [[PRED_UDIV_CONTINUE10:%.*]]
-; CHECK: pred.udiv.if9:
-; CHECK-NEXT: [[TMP22:%.*]] = extractelement <32 x i32> [[TMP1]], i64 5
-; CHECK-NEXT: [[TMP23:%.*]] = udiv i32 [[TMP22]], [[D]]
-; CHECK-NEXT: [[TMP24:%.*]] = insertelement <32 x i32> [[TMP21]], i32 [[TMP23]], i64 5
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE10]]
-; CHECK: pred.udiv.continue10:
-; CHECK-NEXT: [[TMP25:%.*]] = phi <32 x i32> [ [[TMP21]], [[PRED_UDIV_CONTINUE8]] ], [ [[TMP24]], [[PRED_UDIV_IF9]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF11:%.*]], label [[PRED_UDIV_CONTINUE12:%.*]]
-; CHECK: pred.udiv.if11:
-; CHECK-NEXT: [[TMP26:%.*]] = extractelement <32 x i32> [[TMP1]], i64 6
-; CHECK-NEXT: [[TMP27:%.*]] = udiv i32 [[TMP26]], [[D]]
-; CHECK-NEXT: [[TMP28:%.*]] = insertelement <32 x i32> [[TMP25]], i32 [[TMP27]], i64 6
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE12]]
-; CHECK: pred.udiv.continue12:
-; CHECK-NEXT: [[TMP29:%.*]] = phi <32 x i32> [ [[TMP25]], [[PRED_UDIV_CONTINUE10]] ], [ [[TMP28]], [[PRED_UDIV_IF11]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF13:%.*]], label [[PRED_UDIV_CONTINUE14:%.*]]
-; CHECK: pred.udiv.if13:
-; CHECK-NEXT: [[TMP30:%.*]] = extractelement <32 x i32> [[TMP1]], i64 7
-; CHECK-NEXT: [[TMP31:%.*]] = udiv i32 [[TMP30]], [[D]]
-; CHECK-NEXT: [[TMP32:%.*]] = insertelement <32 x i32> [[TMP29]], i32 [[TMP31]], i64 7
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE14]]
-; CHECK: pred.udiv.continue14:
-; CHECK-NEXT: [[TMP33:%.*]] = phi <32 x i32> [ [[TMP29]], [[PRED_UDIV_CONTINUE12]] ], [ [[TMP32]], [[PRED_UDIV_IF13]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF15:%.*]], label [[PRED_UDIV_CONTINUE16:%.*]]
-; CHECK: pred.udiv.if15:
-; CHECK-NEXT: [[TMP34:%.*]] = extractelement <32 x i32> [[TMP1]], i64 8
-; CHECK-NEXT: [[TMP35:%.*]] = udiv i32 [[TMP34]], [[D]]
-; CHECK-NEXT: [[TMP36:%.*]] = insertelement <32 x i32> [[TMP33]], i32 [[TMP35]], i64 8
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE16]]
-; CHECK: pred.udiv.continue16:
-; CHECK-NEXT: [[TMP37:%.*]] = phi <32 x i32> [ [[TMP33]], [[PRED_UDIV_CONTINUE14]] ], [ [[TMP36]], [[PRED_UDIV_IF15]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF17:%.*]], label [[PRED_UDIV_CONTINUE18:%.*]]
-; CHECK: pred.udiv.if17:
-; CHECK-NEXT: [[TMP38:%.*]] = extractelement <32 x i32> [[TMP1]], i64 9
-; CHECK-NEXT: [[TMP39:%.*]] = udiv i32 [[TMP38]], [[D]]
-; CHECK-NEXT: [[TMP40:%.*]] = insertelement <32 x i32> [[TMP37]], i32 [[TMP39]], i64 9
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE18]]
-; CHECK: pred.udiv.continue18:
-; CHECK-NEXT: [[TMP41:%.*]] = phi <32 x i32> [ [[TMP37]], [[PRED_UDIV_CONTINUE16]] ], [ [[TMP40]], [[PRED_UDIV_IF17]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF19:%.*]], label [[PRED_UDIV_CONTINUE20:%.*]]
-; CHECK: pred.udiv.if19:
-; CHECK-NEXT: [[TMP42:%.*]] = extractelement <32 x i32> [[TMP1]], i64 10
-; CHECK-NEXT: [[TMP43:%.*]] = udiv i32 [[TMP42]], [[D]]
-; CHECK-NEXT: [[TMP44:%.*]] = insertelement <32 x i32> [[TMP41]], i32 [[TMP43]], i64 10
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE20]]
-; CHECK: pred.udiv.continue20:
-; CHECK-NEXT: [[TMP45:%.*]] = phi <32 x i32> [ [[TMP41]], [[PRED_UDIV_CONTINUE18]] ], [ [[TMP44]], [[PRED_UDIV_IF19]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF21:%.*]], label [[PRED_UDIV_CONTINUE22:%.*]]
-; CHECK: pred.udiv.if21:
-; CHECK-NEXT: [[TMP46:%.*]] = extractelement <32 x i32> [[TMP1]], i64 11
-; CHECK-NEXT: [[TMP47:%.*]] = udiv i32 [[TMP46]], [[D]]
-; CHECK-NEXT: [[TMP48:%.*]] = insertelement <32 x i32> [[TMP45]], i32 [[TMP47]], i64 11
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE22]]
-; CHECK: pred.udiv.continue22:
-; CHECK-NEXT: [[TMP49:%.*]] = phi <32 x i32> [ [[TMP45]], [[PRED_UDIV_CONTINUE20]] ], [ [[TMP48]], [[PRED_UDIV_IF21]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF23:%.*]], label [[PRED_UDIV_CONTINUE24:%.*]]
-; CHECK: pred.udiv.if23:
-; CHECK-NEXT: [[TMP50:%.*]] = extractelement <32 x i32> [[TMP1]], i64 12
-; CHECK-NEXT: [[TMP51:%.*]] = udiv i32 [[TMP50]], [[D]]
-; CHECK-NEXT: [[TMP52:%.*]] = insertelement <32 x i32> [[TMP49]], i32 [[TMP51]], i64 12
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE24]]
-; CHECK: pred.udiv.continue24:
-; CHECK-NEXT: [[TMP53:%.*]] = phi <32 x i32> [ [[TMP49]], [[PRED_UDIV_CONTINUE22]] ], [ [[TMP52]], [[PRED_UDIV_IF23]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF25:%.*]], label [[PRED_UDIV_CONTINUE26:%.*]]
-; CHECK: pred.udiv.if25:
-; CHECK-NEXT: [[TMP54:%.*]] = extractelement <32 x i32> [[TMP1]], i64 13
-; CHECK-NEXT: [[TMP55:%.*]] = udiv i32 [[TMP54]], [[D]]
-; CHECK-NEXT: [[TMP56:%.*]] = insertelement <32 x i32> [[TMP53]], i32 [[TMP55]], i64 13
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE26]]
-; CHECK: pred.udiv.continue26:
-; CHECK-NEXT: [[TMP57:%.*]] = phi <32 x i32> [ [[TMP53]], [[PRED_UDIV_CONTINUE24]] ], [ [[TMP56]], [[PRED_UDIV_IF25]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF27:%.*]], label [[PRED_UDIV_CONTINUE28:%.*]]
-; CHECK: pred.udiv.if27:
-; CHECK-NEXT: [[TMP58:%.*]] = extractelement <32 x i32> [[TMP1]], i64 14
-; CHECK-NEXT: [[TMP59:%.*]] = udiv i32 [[TMP58]], [[D]]
-; CHECK-NEXT: [[TMP60:%.*]] = insertelement <32 x i32> [[TMP57]], i32 [[TMP59]], i64 14
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE28]]
-; CHECK: pred.udiv.continue28:
-; CHECK-NEXT: [[TMP61:%.*]] = phi <32 x i32> [ [[TMP57]], [[PRED_UDIV_CONTINUE26]] ], [ [[TMP60]], [[PRED_UDIV_IF27]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF29:%.*]], label [[PRED_UDIV_CONTINUE30:%.*]]
-; CHECK: pred.udiv.if29:
-; CHECK-NEXT: [[TMP62:%.*]] = extractelement <32 x i32> [[TMP1]], i64 15
-; CHECK-NEXT: [[TMP63:%.*]] = udiv i32 [[TMP62]], [[D]]
-; CHECK-NEXT: [[TMP64:%.*]] = insertelement <32 x i32> [[TMP61]], i32 [[TMP63]], i64 15
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE30]]
-; CHECK: pred.udiv.continue30:
-; CHECK-NEXT: [[TMP65:%.*]] = phi <32 x i32> [ [[TMP61]], [[PRED_UDIV_CONTINUE28]] ], [ [[TMP64]], [[PRED_UDIV_IF29]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF31:%.*]], label [[PRED_UDIV_CONTINUE32:%.*]]
-; CHECK: pred.udiv.if31:
-; CHECK-NEXT: [[TMP66:%.*]] = extractelement <32 x i32> [[TMP1]], i64 16
-; CHECK-NEXT: [[TMP67:%.*]] = udiv i32 [[TMP66]], [[D]]
-; CHECK-NEXT: [[TMP68:%.*]] = insertelement <32 x i32> [[TMP65]], i32 [[TMP67]], i64 16
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE32]]
-; CHECK: pred.udiv.continue32:
-; CHECK-NEXT: [[TMP69:%.*]] = phi <32 x i32> [ [[TMP65]], [[PRED_UDIV_CONTINUE30]] ], [ [[TMP68]], [[PRED_UDIV_IF31]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF33:%.*]], label [[PRED_UDIV_CONTINUE34:%.*]]
-; CHECK: pred.udiv.if33:
-; CHECK-NEXT: [[TMP70:%.*]] = extractelement <32 x i32> [[TMP1]], i64 17
-; CHECK-NEXT: [[TMP71:%.*]] = udiv i32 [[TMP70]], [[D]]
-; CHECK-NEXT: [[TMP72:%.*]] = insertelement <32 x i32> [[TMP69]], i32 [[TMP71]], i64 17
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE34]]
-; CHECK: pred.udiv.continue34:
-; CHECK-NEXT: [[TMP73:%.*]] = phi <32 x i32> [ [[TMP69]], [[PRED_UDIV_CONTINUE32]] ], [ [[TMP72]], [[PRED_UDIV_IF33]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF35:%.*]], label [[PRED_UDIV_CONTINUE36:%.*]]
-; CHECK: pred.udiv.if35:
-; CHECK-NEXT: [[TMP74:%.*]] = extractelement <32 x i32> [[TMP1]], i64 18
-; CHECK-NEXT: [[TMP75:%.*]] = udiv i32 [[TMP74]], [[D]]
-; CHECK-NEXT: [[TMP76:%.*]] = insertelement <32 x i32> [[TMP73]], i32 [[TMP75]], i64 18
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE36]]
-; CHECK: pred.udiv.continue36:
-; CHECK-NEXT: [[TMP77:%.*]] = phi <32 x i32> [ [[TMP73]], [[PRED_UDIV_CONTINUE34]] ], [ [[TMP76]], [[PRED_UDIV_IF35]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF37:%.*]], label [[PRED_UDIV_CONTINUE38:%.*]]
-; CHECK: pred.udiv.if37:
-; CHECK-NEXT: [[TMP78:%.*]] = extractelement <32 x i32> [[TMP1]], i64 19
-; CHECK-NEXT: [[TMP79:%.*]] = udiv i32 [[TMP78]], [[D]]
-; CHECK-NEXT: [[TMP80:%.*]] = insertelement <32 x i32> [[TMP77]], i32 [[TMP79]], i64 19
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE38]]
-; CHECK: pred.udiv.continue38:
-; CHECK-NEXT: [[TMP81:%.*]] = phi <32 x i32> [ [[TMP77]], [[PRED_UDIV_CONTINUE36]] ], [ [[TMP80]], [[PRED_UDIV_IF37]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF39:%.*]], label [[PRED_UDIV_CONTINUE40:%.*]]
-; CHECK: pred.udiv.if39:
-; CHECK-NEXT: [[TMP82:%.*]] = extractelement <32 x i32> [[TMP1]], i64 20
-; CHECK-NEXT: [[TMP83:%.*]] = udiv i32 [[TMP82]], [[D]]
-; CHECK-NEXT: [[TMP84:%.*]] = insertelement <32 x i32> [[TMP81]], i32 [[TMP83]], i64 20
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE40]]
-; CHECK: pred.udiv.continue40:
-; CHECK-NEXT: [[TMP85:%.*]] = phi <32 x i32> [ [[TMP81]], [[PRED_UDIV_CONTINUE38]] ], [ [[TMP84]], [[PRED_UDIV_IF39]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF41:%.*]], label [[PRED_UDIV_CONTINUE42:%.*]]
-; CHECK: pred.udiv.if41:
-; CHECK-NEXT: [[TMP86:%.*]] = extractelement <32 x i32> [[TMP1]], i64 21
-; CHECK-NEXT: [[TMP87:%.*]] = udiv i32 [[TMP86]], [[D]]
-; CHECK-NEXT: [[TMP88:%.*]] = insertelement <32 x i32> [[TMP85]], i32 [[TMP87]], i64 21
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE42]]
-; CHECK: pred.udiv.continue42:
-; CHECK-NEXT: [[TMP89:%.*]] = phi <32 x i32> [ [[TMP85]], [[PRED_UDIV_CONTINUE40]] ], [ [[TMP88]], [[PRED_UDIV_IF41]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF43:%.*]], label [[PRED_UDIV_CONTINUE44:%.*]]
-; CHECK: pred.udiv.if43:
-; CHECK-NEXT: [[TMP90:%.*]] = extractelement <32 x i32> [[TMP1]], i64 22
-; CHECK-NEXT: [[TMP91:%.*]] = udiv i32 [[TMP90]], [[D]]
-; CHECK-NEXT: [[TMP92:%.*]] = insertelement <32 x i32> [[TMP89]], i32 [[TMP91]], i64 22
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE44]]
-; CHECK: pred.udiv.continue44:
-; CHECK-NEXT: [[TMP93:%.*]] = phi <32 x i32> [ [[TMP89]], [[PRED_UDIV_CONTINUE42]] ], [ [[TMP92]], [[PRED_UDIV_IF43]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF45:%.*]], label [[PRED_UDIV_CONTINUE46:%.*]]
-; CHECK: pred.udiv.if45:
-; CHECK-NEXT: [[TMP94:%.*]] = extractelement <32 x i32> [[TMP1]], i64 23
-; CHECK-NEXT: [[TMP95:%.*]] = udiv i32 [[TMP94]], [[D]]
-; CHECK-NEXT: [[TMP96:%.*]] = insertelement <32 x i32> [[TMP93]], i32 [[TMP95]], i64 23
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE46]]
-; CHECK: pred.udiv.continue46:
-; CHECK-NEXT: [[TMP97:%.*]] = phi <32 x i32> [ [[TMP93]], [[PRED_UDIV_CONTINUE44]] ], [ [[TMP96]], [[PRED_UDIV_IF45]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF47:%.*]], label [[PRED_UDIV_CONTINUE48:%.*]]
-; CHECK: pred.udiv.if47:
-; CHECK-NEXT: [[TMP98:%.*]] = extractelement <32 x i32> [[TMP1]], i64 24
-; CHECK-NEXT: [[TMP99:%.*]] = udiv i32 [[TMP98]], [[D]]
-; CHECK-NEXT: [[TMP100:%.*]] = insertelement <32 x i32> [[TMP97]], i32 [[TMP99]], i64 24
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE48]]
-; CHECK: pred.udiv.continue48:
-; CHECK-NEXT: [[TMP101:%.*]] = phi <32 x i32> [ [[TMP97]], [[PRED_UDIV_CONTINUE46]] ], [ [[TMP100]], [[PRED_UDIV_IF47]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF49:%.*]], label [[PRED_UDIV_CONTINUE50:%.*]]
-; CHECK: pred.udiv.if49:
-; CHECK-NEXT: [[TMP102:%.*]] = extractelement <32 x i32> [[TMP1]], i64 25
-; CHECK-NEXT: [[TMP103:%.*]] = udiv i32 [[TMP102]], [[D]]
-; CHECK-NEXT: [[TMP104:%.*]] = insertelement <32 x i32> [[TMP101]], i32 [[TMP103]], i64 25
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE50]]
-; CHECK: pred.udiv.continue50:
-; CHECK-NEXT: [[TMP105:%.*]] = phi <32 x i32> [ [[TMP101]], [[PRED_UDIV_CONTINUE48]] ], [ [[TMP104]], [[PRED_UDIV_IF49]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF51:%.*]], label [[PRED_UDIV_CONTINUE52:%.*]]
-; CHECK: pred.udiv.if51:
-; CHECK-NEXT: [[TMP106:%.*]] = extractelement <32 x i32> [[TMP1]], i64 26
-; CHECK-NEXT: [[TMP107:%.*]] = udiv i32 [[TMP106]], [[D]]
-; CHECK-NEXT: [[TMP108:%.*]] = insertelement <32 x i32> [[TMP105]], i32 [[TMP107]], i64 26
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE52]]
-; CHECK: pred.udiv.continue52:
-; CHECK-NEXT: [[TMP109:%.*]] = phi <32 x i32> [ [[TMP105]], [[PRED_UDIV_CONTINUE50]] ], [ [[TMP108]], [[PRED_UDIV_IF51]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF53:%.*]], label [[PRED_UDIV_CONTINUE54:%.*]]
-; CHECK: pred.udiv.if53:
-; CHECK-NEXT: [[TMP110:%.*]] = extractelement <32 x i32> [[TMP1]], i64 27
-; CHECK-NEXT: [[TMP111:%.*]] = udiv i32 [[TMP110]], [[D]]
-; CHECK-NEXT: [[TMP112:%.*]] = insertelement <32 x i32> [[TMP109]], i32 [[TMP111]], i64 27
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE54]]
-; CHECK: pred.udiv.continue54:
-; CHECK-NEXT: [[TMP113:%.*]] = phi <32 x i32> [ [[TMP109]], [[PRED_UDIV_CONTINUE52]] ], [ [[TMP112]], [[PRED_UDIV_IF53]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF55:%.*]], label [[PRED_UDIV_CONTINUE56:%.*]]
-; CHECK: pred.udiv.if55:
-; CHECK-NEXT: [[TMP114:%.*]] = extractelement <32 x i32> [[TMP1]], i64 28
-; CHECK-NEXT: [[TMP115:%.*]] = udiv i32 [[TMP114]], [[D]]
-; CHECK-NEXT: [[TMP116:%.*]] = insertelement <32 x i32> [[TMP113]], i32 [[TMP115]], i64 28
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE56]]
-; CHECK: pred.udiv.continue56:
-; CHECK-NEXT: [[TMP117:%.*]] = phi <32 x i32> [ [[TMP113]], [[PRED_UDIV_CONTINUE54]] ], [ [[TMP116]], [[PRED_UDIV_IF55]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF57:%.*]], label [[PRED_UDIV_CONTINUE58:%.*]]
-; CHECK: pred.udiv.if57:
-; CHECK-NEXT: [[TMP118:%.*]] = extractelement <32 x i32> [[TMP1]], i64 29
-; CHECK-NEXT: [[TMP119:%.*]] = udiv i32 [[TMP118]], [[D]]
-; CHECK-NEXT: [[TMP120:%.*]] = insertelement <32 x i32> [[TMP117]], i32 [[TMP119]], i64 29
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE58]]
-; CHECK: pred.udiv.continue58:
-; CHECK-NEXT: [[TMP121:%.*]] = phi <32 x i32> [ [[TMP117]], [[PRED_UDIV_CONTINUE56]] ], [ [[TMP120]], [[PRED_UDIV_IF57]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF59:%.*]], label [[PRED_UDIV_CONTINUE60:%.*]]
-; CHECK: pred.udiv.if59:
-; CHECK-NEXT: [[TMP122:%.*]] = extractelement <32 x i32> [[TMP1]], i64 30
-; CHECK-NEXT: [[TMP123:%.*]] = udiv i32 [[TMP122]], [[D]]
-; CHECK-NEXT: [[TMP124:%.*]] = insertelement <32 x i32> [[TMP121]], i32 [[TMP123]], i64 30
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE60]]
-; CHECK: pred.udiv.continue60:
-; CHECK-NEXT: [[TMP125:%.*]] = phi <32 x i32> [ [[TMP121]], [[PRED_UDIV_CONTINUE58]] ], [ [[TMP124]], [[PRED_UDIV_IF59]] ]
-; CHECK-NEXT: br i1 [[TMP0]], label [[PRED_UDIV_IF61:%.*]], label [[PRED_UDIV_CONTINUE62]]
-; CHECK: pred.udiv.if61:
-; CHECK-NEXT: [[TMP126:%.*]] = extractelement <32 x i32> [[TMP1]], i64 31
-; CHECK-NEXT: [[TMP127:%.*]] = udiv i32 [[TMP126]], [[D]]
-; CHECK-NEXT: [[TMP128:%.*]] = insertelement <32 x i32> [[TMP125]], i32 [[TMP127]], i64 31
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE62]]
-; CHECK: pred.udiv.continue62:
-; CHECK-NEXT: [[TMP129:%.*]] = phi <32 x i32> [ [[TMP125]], [[PRED_UDIV_CONTINUE60]] ], [ [[TMP128]], [[PRED_UDIV_IF61]] ]
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 32
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add <32 x i32> [[VEC_IND]], splat (i32 32)
-; CHECK-NEXT: [[TMP130:%.*]] = icmp eq i32 [[INDEX_NEXT]], 992
-; CHECK-NEXT: br i1 [[TMP130]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
-; CHECK: middle.block:
-; CHECK-NEXT: [[TMP131:%.*]] = zext <32 x i32> [[TMP129]] to <32 x i64>
-; CHECK-NEXT: [[PREDPHI:%.*]] = select i1 [[C]], <32 x i64> zeroinitializer, <32 x i64> [[TMP131]]
-; CHECK-NEXT: [[TMP132:%.*]] = extractelement <32 x i64> [[PREDPHI]], i64 31
-; CHECK-NEXT: br i1 false, label [[EXIT:%.*]], label [[VEC_EPILOG_ITER_CHECK:%.*]]
-; CHECK: vec.epilog.iter.check:
-; CHECK-NEXT: br i1 false, label [[VEC_EPILOG_SCALAR_PH]], label [[VEC_EPILOG_PH]], !prof [[PROF13:![0-9]+]]
-; CHECK: vec.epilog.ph:
-; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ 992, [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT: [[TMP133:%.*]] = xor i1 [[C]], true
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i32> poison, i32 [[VEC_EPILOG_RESUME_VAL]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT]], <8 x i32> poison, <8 x i32> zeroinitializer
-; CHECK-NEXT: [[INDUCTION:%.*]] = add <8 x i32> [[BROADCAST_SPLAT]], <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-; CHECK-NEXT: br label [[VEC_EPILOG_VECTOR_BODY:%.*]]
-; CHECK: vec.epilog.vector.body:
-; CHECK-NEXT: [[INDEX63:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], [[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT81:%.*]], [[PRED_UDIV_CONTINUE80:%.*]] ]
-; CHECK-NEXT: [[VEC_IND64:%.*]] = phi <8 x i32> [ [[INDUCTION]], [[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT82:%.*]], [[PRED_UDIV_CONTINUE80]] ]
-; CHECK-NEXT: [[TMP134:%.*]] = call <8 x i32> @llvm.usub.sat.v8i32(<8 x i32> [[VEC_IND64]], <8 x i32> splat (i32 1))
-; CHECK-NEXT: br i1 [[TMP133]], label [[PRED_UDIV_IF65:%.*]], label [[PRED_UDIV_CONTINUE66:%.*]]
-; CHECK: pred.udiv.if65:
-; CHECK-NEXT: [[TMP135:%.*]] = extractelement <8 x i32> [[TMP134]], i64 0
-; CHECK-NEXT: [[TMP136:%.*]] = udiv i32 [[TMP135]], [[D]]
-; CHECK-NEXT: [[TMP137:%.*]] = insertelement <8 x i32> poison, i32 [[TMP136]], i64 0
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE66]]
-; CHECK: pred.udiv.continue66:
-; CHECK-NEXT: [[TMP138:%.*]] = phi <8 x i32> [ poison, [[VEC_EPILOG_VECTOR_BODY]] ], [ [[TMP137]], [[PRED_UDIV_IF65]] ]
-; CHECK-NEXT: br i1 [[TMP133]], label [[PRED_UDIV_IF67:%.*]], label [[PRED_UDIV_CONTINUE68:%.*]]
-; CHECK: pred.udiv.if67:
-; CHECK-NEXT: [[TMP139:%.*]] = extractelement <8 x i32> [[TMP134]], i64 1
-; CHECK-NEXT: [[TMP140:%.*]] = udiv i32 [[TMP139]], [[D]]
-; CHECK-NEXT: [[TMP141:%.*]] = insertelement <8 x i32> [[TMP138]], i32 [[TMP140]], i64 1
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE68]]
-; CHECK: pred.udiv.continue68:
-; CHECK-NEXT: [[TMP142:%.*]] = phi <8 x i32> [ [[TMP138]], [[PRED_UDIV_CONTINUE66]] ], [ [[TMP141]], [[PRED_UDIV_IF67]] ]
-; CHECK-NEXT: br i1 [[TMP133]], label [[PRED_UDIV_IF69:%.*]], label [[PRED_UDIV_CONTINUE70:%.*]]
-; CHECK: pred.udiv.if69:
-; CHECK-NEXT: [[TMP143:%.*]] = extractelement <8 x i32> [[TMP134]], i64 2
-; CHECK-NEXT: [[TMP144:%.*]] = udiv i32 [[TMP143]], [[D]]
-; CHECK-NEXT: [[TMP145:%.*]] = insertelement <8 x i32> [[TMP142]], i32 [[TMP144]], i64 2
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE70]]
-; CHECK: pred.udiv.continue70:
-; CHECK-NEXT: [[TMP146:%.*]] = phi <8 x i32> [ [[TMP142]], [[PRED_UDIV_CONTINUE68]] ], [ [[TMP145]], [[PRED_UDIV_IF69]] ]
-; CHECK-NEXT: br i1 [[TMP133]], label [[PRED_UDIV_IF71:%.*]], label [[PRED_UDIV_CONTINUE72:%.*]]
-; CHECK: pred.udiv.if71:
-; CHECK-NEXT: [[TMP147:%.*]] = extractelement <8 x i32> [[TMP134]], i64 3
-; CHECK-NEXT: [[TMP148:%.*]] = udiv i32 [[TMP147]], [[D]]
-; CHECK-NEXT: [[TMP149:%.*]] = insertelement <8 x i32> [[TMP146]], i32 [[TMP148]], i64 3
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE72]]
-; CHECK: pred.udiv.continue72:
-; CHECK-NEXT: [[TMP150:%.*]] = phi <8 x i32> [ [[TMP146]], [[PRED_UDIV_CONTINUE70]] ], [ [[TMP149]], [[PRED_UDIV_IF71]] ]
-; CHECK-NEXT: br i1 [[TMP133]], label [[PRED_UDIV_IF73:%.*]], label [[PRED_UDIV_CONTINUE74:%.*]]
-; CHECK: pred.udiv.if73:
-; CHECK-NEXT: [[TMP151:%.*]] = extractelement <8 x i32> [[TMP134]], i64 4
-; CHECK-NEXT: [[TMP152:%.*]] = udiv i32 [[TMP151]], [[D]]
-; CHECK-NEXT: [[TMP153:%.*]] = insertelement <8 x i32> [[TMP150]], i32 [[TMP152]], i64 4
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE74]]
-; CHECK: pred.udiv.continue74:
-; CHECK-NEXT: [[TMP154:%.*]] = phi <8 x i32> [ [[TMP150]], [[PRED_UDIV_CONTINUE72]] ], [ [[TMP153]], [[PRED_UDIV_IF73]] ]
-; CHECK-NEXT: br i1 [[TMP133]], label [[PRED_UDIV_IF75:%.*]], label [[PRED_UDIV_CONTINUE76:%.*]]
-; CHECK: pred.udiv.if75:
-; CHECK-NEXT: [[TMP155:%.*]] = extractelement <8 x i32> [[TMP134]], i64 5
-; CHECK-NEXT: [[TMP156:%.*]] = udiv i32 [[TMP155]], [[D]]
-; CHECK-NEXT: [[TMP157:%.*]] = insertelement <8 x i32> [[TMP154]], i32 [[TMP156]], i64 5
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE76]]
-; CHECK: pred.udiv.continue76:
-; CHECK-NEXT: [[TMP158:%.*]] = phi <8 x i32> [ [[TMP154]], [[PRED_UDIV_CONTINUE74]] ], [ [[TMP157]], [[PRED_UDIV_IF75]] ]
-; CHECK-NEXT: br i1 [[TMP133]], label [[PRED_UDIV_IF77:%.*]], label [[PRED_UDIV_CONTINUE78:%.*]]
-; CHECK: pred.udiv.if77:
-; CHECK-NEXT: [[TMP159:%.*]] = extractelement <8 x i32> [[TMP134]], i64 6
-; CHECK-NEXT: [[TMP160:%.*]] = udiv i32 [[TMP159]], [[D]]
-; CHECK-NEXT: [[TMP161:%.*]] = insertelement <8 x i32> [[TMP158]], i32 [[TMP160]], i64 6
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE78]]
-; CHECK: pred.udiv.continue78:
-; CHECK-NEXT: [[TMP162:%.*]] = phi <8 x i32> [ [[TMP158]], [[PRED_UDIV_CONTINUE76]] ], [ [[TMP161]], [[PRED_UDIV_IF77]] ]
-; CHECK-NEXT: br i1 [[TMP133]], label [[PRED_UDIV_IF79:%.*]], label [[PRED_UDIV_CONTINUE80]]
-; CHECK: pred.udiv.if79:
-; CHECK-NEXT: [[TMP163:%.*]] = extractelement <8 x i32> [[TMP134]], i64 7
-; CHECK-NEXT: [[TMP164:%.*]] = udiv i32 [[TMP163]], [[D]]
-; CHECK-NEXT: [[TMP165:%.*]] = insertelement <8 x i32> [[TMP162]], i32 [[TMP164]], i64 7
-; CHECK-NEXT: br label [[PRED_UDIV_CONTINUE80]]
-; CHECK: pred.udiv.continue80:
-; CHECK-NEXT: [[TMP166:%.*]] = phi <8 x i32> [ [[TMP162]], [[PRED_UDIV_CONTINUE78]] ], [ [[TMP165]], [[PRED_UDIV_IF79]] ]
-; CHECK-NEXT: [[INDEX_NEXT81]] = add nuw i32 [[INDEX63]], 8
-; CHECK-NEXT: [[VEC_IND_NEXT82]] = add <8 x i32> [[VEC_IND64]], splat (i32 8)
-; CHECK-NEXT: [[TMP167:%.*]] = icmp eq i32 [[INDEX_NEXT81]], 1000
-; CHECK-NEXT: br i1 [[TMP167]], label [[VEC_EPILOG_MIDDLE_BLOCK:%.*]], label [[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP14:![0-9]+]]
-; CHECK: vec.epilog.middle.block:
-; CHECK-NEXT: [[TMP168:%.*]] = zext <8 x i32> [[TMP166]] to <8 x i64>
-; CHECK-NEXT: [[PREDPHI83:%.*]] = select i1 [[C]], <8 x i64> zeroinitializer, <8 x i64> [[TMP168]]
-; CHECK-NEXT: [[TMP169:%.*]] = extractelement <8 x i64> [[PREDPHI83]], i64 7
-; CHECK-NEXT: br i1 false, label [[EXIT]], label [[VEC_EPILOG_SCALAR_PH]]
-; CHECK: vec.epilog.scalar.ph:
-; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ 1000, [[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 992, [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[ITER_CHECK:%.*]] ]
+; CHECK-NEXT: entry:
; CHECK-NEXT: br label [[LOOP_HEADER:%.*]]
; CHECK: loop.header:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], [[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], [[LOOP_LATCH:%.*]] ]
-; CHECK-NEXT: br i1 [[C]], label [[LOOP_LATCH]], label [[THEN:%.*]]
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[LOOP_LATCH:%.*]] ]
+; CHECK-NEXT: br i1 [[C:%.*]], label [[LOOP_LATCH]], label [[THEN:%.*]]
; CHECK: then:
; CHECK-NEXT: [[CALL:%.*]] = tail call i32 @llvm.usub.sat.i32(i32 [[IV]], i32 1)
-; CHECK-NEXT: [[UDIV:%.*]] = udiv i32 [[CALL]], [[D]]
+; CHECK-NEXT: [[UDIV:%.*]] = udiv i32 [[CALL]], [[D:%.*]]
; CHECK-NEXT: [[ZEXT:%.*]] = zext i32 [[UDIV]] to i64
; CHECK-NEXT: br label [[LOOP_LATCH]]
; CHECK: loop.latch:
; CHECK-NEXT: [[MERGE:%.*]] = phi i64 [ [[ZEXT]], [[THEN]] ], [ 0, [[LOOP_HEADER]] ]
; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
; CHECK-NEXT: [[EC:%.*]] = icmp eq i32 [[IV]], 1000
-; CHECK-NEXT: br i1 [[EC]], label [[EXIT]], label [[LOOP_HEADER]], !llvm.loop [[LOOP15:![0-9]+]]
+; CHECK-NEXT: br i1 [[EC]], label [[EXIT:%.*]], label [[LOOP_HEADER]]
; CHECK: exit:
-; CHECK-NEXT: [[MERGE_LCSSA:%.*]] = phi i64 [ [[MERGE]], [[LOOP_LATCH]] ], [ [[TMP132]], [[MIDDLE_BLOCK]] ], [ [[TMP169]], [[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-NEXT: [[MERGE_LCSSA:%.*]] = phi i64 [ [[MERGE]], [[LOOP_LATCH]] ]
; CHECK-NEXT: ret i64 [[MERGE_LCSSA]]
;
entry:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll b/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll
index b75d41907ee90..a1f46e7d41d64 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/cost-model.ll
@@ -454,7 +454,7 @@ define void @multi_exit(ptr %dst, ptr %src.1, ptr %src.2, i64 %A, i64 %B) #0 {
; CHECK-NEXT: [[TMP1:%.*]] = freeze i64 [[TMP0]]
; CHECK-NEXT: [[UMIN10:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP1]], i64 [[A]])
; CHECK-NEXT: [[TMP2:%.*]] = add nuw i64 [[UMIN10]], 1
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ule i64 [[TMP2]], 24
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ule i64 [[TMP2]], 32
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
; CHECK: [[VECTOR_SCEVCHECK]]:
; CHECK-NEXT: [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[B]], i64 1)
@@ -490,31 +490,31 @@ define void @multi_exit(ptr %dst, ptr %src.1, ptr %src.2, i64 %A, i64 %B) #0 {
; CHECK-NEXT: [[CONFLICT_RDX:%.*]] = or i1 [[FOUND_CONFLICT]], [[FOUND_CONFLICT8]]
; CHECK-NEXT: br i1 [[CONFLICT_RDX]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i64 [[TMP2]], 3
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i64 [[TMP2]], 31
; CHECK-NEXT: [[TMP19:%.*]] = icmp eq i64 [[N_MOD_VF]], 0
-; CHECK-NEXT: [[TMP20:%.*]] = select i1 [[TMP19]], i64 4, i64 [[N_MOD_VF]]
+; CHECK-NEXT: [[TMP20:%.*]] = select i1 [[TMP19]], i64 32, i64 [[N_MOD_VF]]
; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP2]], [[TMP20]]
; CHECK-NEXT: [[TMP21:%.*]] = trunc i64 [[N_VEC]] to i32
; CHECK-NEXT: [[TMP22:%.*]] = load i64, ptr [[SRC_2]], align 8, !alias.scope [[META6:![0-9]+]]
; CHECK-NEXT: [[TMP23:%.*]] = icmp ne i64 [[TMP22]], 0
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <2 x i1> poison, i1 [[TMP23]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <2 x i1> [[BROADCAST_SPLATINSERT]], <2 x i1> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <16 x i1> poison, i1 [[TMP23]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <16 x i1> [[BROADCAST_SPLATINSERT]], <16 x i1> poison, <16 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP24:%.*]] = trunc i64 [[INDEX]] to i32
; CHECK-NEXT: [[TMP25:%.*]] = getelementptr inbounds i64, ptr [[SRC_1]], i32 [[TMP24]]
-; CHECK-NEXT: [[TMP26:%.*]] = getelementptr inbounds i64, ptr [[TMP25]], i64 2
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x i64>, ptr [[TMP26]], align 8, !alias.scope [[META9:![0-9]+]]
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP26:%.*]] = getelementptr inbounds i64, ptr [[TMP25]], i64 16
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i64>, ptr [[TMP26]], align 8, !alias.scope [[META9:![0-9]+]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; CHECK-NEXT: [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP31]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: [[TMP27:%.*]] = icmp eq <2 x i64> [[WIDE_LOAD]], zeroinitializer
-; CHECK-NEXT: [[TMP28:%.*]] = and <2 x i1> [[BROADCAST_SPLAT]], [[TMP27]]
-; CHECK-NEXT: [[TMP29:%.*]] = zext <2 x i1> [[TMP28]] to <2 x i8>
-; CHECK-NEXT: [[TMP30:%.*]] = extractelement <2 x i8> [[TMP29]], i64 1
-; CHECK-NEXT: store i8 [[TMP30]], ptr [[DST]], align 1, !alias.scope [[META12:![0-9]+]], !noalias [[META14:![0-9]+]]
+; CHECK-NEXT: [[TMP28:%.*]] = icmp eq <16 x i64> [[WIDE_LOAD]], zeroinitializer
+; CHECK-NEXT: [[TMP29:%.*]] = and <16 x i1> [[BROADCAST_SPLAT]], [[TMP28]]
+; CHECK-NEXT: [[TMP30:%.*]] = zext <16 x i1> [[TMP29]] to <16 x i8>
+; CHECK-NEXT: [[TMP32:%.*]] = extractelement <16 x i8> [[TMP30]], i64 15
+; CHECK-NEXT: store i8 [[TMP32]], ptr [[DST]], align 1, !alias.scope [[META12:![0-9]+]], !noalias [[META14:![0-9]+]]
; CHECK-NEXT: br label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
;
diff --git a/llvm/test/Transforms/LoopVectorize/X86/epilog-vectorization-inductions.ll b/llvm/test/Transforms/LoopVectorize/X86/epilog-vectorization-inductions.ll
index 3ecab09e93265..b6cd60361827b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/epilog-vectorization-inductions.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/epilog-vectorization-inductions.ll
@@ -246,53 +246,18 @@ exit:
; Test case for https://github.com/llvm/llvm-project/issues/151686.
define i8 @multiple_inductions_start_at_0() {
; CHECK-LABEL: @multiple_inductions_start_at_0(
-; CHECK-NEXT: iter.check:
-; CHECK-NEXT: br i1 false, label [[VEC_EPILOG_SCALAR_PH:%.*]], label [[VECTOR_MAIN_LOOP_ITER_CHECK:%.*]]
-; CHECK: vector.main.loop.iter.check:
-; CHECK-NEXT: br i1 false, label [[VEC_EPILOG_PH:%.*]], label [[VECTOR_PH:%.*]]
-; CHECK: vector.ph:
-; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
-; CHECK: vector.body:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[STEP_ADD_3:%.*]] = phi <32 x i8> [ zeroinitializer, [[VECTOR_PH]] ], [ [[STEP_ADD_3]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 128
-; CHECK-NEXT: [[TMP0:%.*]] = icmp eq i32 [[INDEX_NEXT]], 1024
-; CHECK-NEXT: br i1 [[TMP0]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
-; CHECK: middle.block:
-; CHECK-NEXT: [[TMP1:%.*]] = extractelement <32 x i8> [[STEP_ADD_3]], i64 31
-; CHECK-NEXT: br i1 false, label [[EXIT:%.*]], label [[VEC_EPILOG_ITER_CHECK:%.*]]
-; CHECK: vec.epilog.iter.check:
-; CHECK-NEXT: br i1 false, label [[VEC_EPILOG_SCALAR_PH]], label [[VEC_EPILOG_PH]], !prof [[PROF11:![0-9]+]]
-; CHECK: vec.epilog.ph:
-; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ 1024, [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ 0, [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT: [[TMP2:%.*]] = trunc i32 [[BC_RESUME_VAL]] to i8
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i8> poison, i8 [[TMP2]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i8> [[BROADCAST_SPLATINSERT]], <4 x i8> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: br label [[VEC_EPILOG_VECTOR_BODY:%.*]]
-; CHECK: vec.epilog.vector.body:
-; CHECK-NEXT: [[INDEX1:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], [[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND2:%.*]] = phi <4 x i8> [ [[BROADCAST_SPLAT]], [[VEC_EPILOG_PH]] ], [ [[VEC_IND2]], [[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT: [[INDEX_NEXT3]] = add nuw i32 [[INDEX1]], 4
-; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i32 [[INDEX_NEXT3]], 1052
-; CHECK-NEXT: br i1 [[TMP3]], label [[VEC_EPILOG_MIDDLE_BLOCK:%.*]], label [[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
-; CHECK: vec.epilog.middle.block:
-; CHECK-NEXT: [[TMP4:%.*]] = extractelement <4 x i8> [[VEC_IND2]], i64 3
-; CHECK-NEXT: br i1 true, label [[EXIT]], label [[VEC_EPILOG_SCALAR_PH]]
-; CHECK: vec.epilog.scalar.ph:
-; CHECK-NEXT: [[BC_RESUME_VAL5:%.*]] = phi i32 [ 1052, [[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 1024, [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[ITER_CHECK:%.*]] ]
-; CHECK-NEXT: [[BC_RESUME_VAL6:%.*]] = phi i32 [ -469762048, [[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 0, [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[ITER_CHECK]] ]
+; CHECK-NEXT: entry:
; CHECK-NEXT: br label [[LOOP:%.*]]
; CHECK: loop:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL5]], [[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], [[LOOP]] ]
-; CHECK-NEXT: [[IV_2:%.*]] = phi i32 [ [[BC_RESUME_VAL6]], [[VEC_EPILOG_SCALAR_PH]] ], [ [[ADD:%.*]], [[LOOP]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[LOOP]] ]
+; CHECK-NEXT: [[IV_2:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[ADD:%.*]], [[LOOP]] ]
; CHECK-NEXT: [[ADD]] = add i32 [[IV_2]], -16777216
; CHECK-NEXT: [[TRUNC:%.*]] = trunc i32 [[IV_2]] to i8
; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
; CHECK-NEXT: [[EC:%.*]] = icmp ugt i32 [[IV]], 1050
-; CHECK-NEXT: br i1 [[EC]], label [[EXIT]], label [[LOOP]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK-NEXT: br i1 [[EC]], label [[EXIT:%.*]], label [[LOOP]]
; CHECK: exit:
-; CHECK-NEXT: [[RES:%.*]] = phi i8 [ [[TRUNC]], [[LOOP]] ], [ [[TMP1]], [[MIDDLE_BLOCK]] ], [ [[TMP4]], [[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-NEXT: [[RES:%.*]] = phi i8 [ [[TRUNC]], [[LOOP]] ]
; CHECK-NEXT: ret i8 [[RES]]
;
entry:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/funclet.ll b/llvm/test/Transforms/LoopVectorize/X86/funclet.ll
index fa858d6d6fbbc..201725a95877f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/funclet.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/funclet.ll
@@ -16,19 +16,16 @@ define void @test1() #0 personality ptr @__CxxFrameHandler3 {
; CHECK-NEXT: [[TMP0:%.*]] = catchswitch within none [label %[[CATCH:.*]]] unwind to caller
; CHECK: [[CATCH]]:
; CHECK-NEXT: [[TMP1:%.*]] = catchpad within [[TMP0]] [ptr null, i32 64, ptr null]
-; CHECK-NEXT: br label %[[FOR_BODY:.*]]
-; CHECK: [[FOR_BODY]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[FOR_BODY]] ], [ [[INC:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP2:%.*]] = call <16 x double> @llvm.floor.v16f64(<16 x double> splat (double 1.000000e+00)) [ "funclet"(token [[TMP1]]) ]
-; CHECK-NEXT: [[INC]] = add nuw i32 [[INDEX]], 16
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, %[[CATCH]] ], [ [[INC:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[CALL:%.*]] = call double @floor(double 1.000000e+00) #[[ATTR1:[0-9]+]] [ "funclet"(token [[TMP1]]) ]
+; CHECK-NEXT: [[INC]] = add nuw nsw i32 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i32 [[INC]], 1024
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[TRY_CONT:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK: [[TRY_CONT]]:
-; CHECK-NEXT: br label %[[EXIT:.*]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[VECTOR_BODY]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: store double 1.000000e+00, ptr @sink, align 8
+; CHECK-NEXT: [[CALL_LCSSA:%.*]] = phi double [ [[CALL]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: store double [[CALL_LCSSA]], ptr @sink, align 8
; CHECK-NEXT: catchret from [[TMP1]] to label %[[TRY_CONT1:.*]]
; CHECK: [[TRY_CONT1]]:
; CHECK-NEXT: ret void
diff --git a/llvm/test/Transforms/LoopVectorize/X86/gcc-examples.ll b/llvm/test/Transforms/LoopVectorize/X86/gcc-examples.ll
index 05e855bc338d2..c9f1afcc61e4c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/gcc-examples.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/gcc-examples.ll
@@ -1,3 +1,4 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
; RUN: opt < %s -passes=loop-vectorize -mtriple=x86_64-apple-macosx10.8.0 -mcpu=corei7 -S | FileCheck %s
; RUN: opt < %s -passes=loop-vectorize -mtriple=x86_64-apple-macosx10.8.0 -mcpu=corei7 -force-vector-interleave=0 -S | FileCheck %s -check-prefix=UNROLL
@@ -9,21 +10,66 @@ target triple = "x86_64-apple-macosx10.8.0"
@a = common global [2048 x i32] zeroinitializer, align 16
; Select VF = 8;
-;CHECK-LABEL: @example1(
-;CHECK: load <4 x i32>
-;CHECK: add nsw <4 x i32>
-;CHECK: store <4 x i32>
-;CHECK: ret void
-;UNROLL-LABEL: @example1(
-;UNROLL: load <4 x i32>
-;UNROLL: load <4 x i32>
-;UNROLL: add nsw <4 x i32>
-;UNROLL: add nsw <4 x i32>
-;UNROLL: store <4 x i32>
-;UNROLL: store <4 x i32>
-;UNROLL: ret void
define void @example1() {
+; CHECK-LABEL: define void @example1(
+; CHECK-SAME: ) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds [2048 x i32], ptr @b, i64 0, i64 [[INDEX]]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 4
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds [2048 x i32], ptr @c, i64 0, i64 [[INDEX]]
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP3]], i64 4
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT: [[TMP5:%.*]] = add nsw <4 x i32> [[WIDE_LOAD2]], [[WIDE_LOAD]]
+; CHECK-NEXT: [[TMP6:%.*]] = add nsw <4 x i32> [[WIDE_LOAD3]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds [2048 x i32], ptr @a, i64 0, i64 [[INDEX]]
+; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 4
+; CHECK-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP7]], align 4
+; CHECK-NEXT: store <4 x i32> [[TMP6]], ptr [[TMP8]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256
+; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[BB10:.*]]
+; CHECK: [[BB10]]:
+; CHECK-NEXT: ret void
+;
+; UNROLL-LABEL: define void @example1(
+; UNROLL-SAME: ) #[[ATTR0:[0-9]+]] {
+; UNROLL-NEXT: br label %[[VECTOR_PH:.*]]
+; UNROLL: [[VECTOR_PH]]:
+; UNROLL-NEXT: br label %[[VECTOR_BODY:.*]]
+; UNROLL: [[VECTOR_BODY]]:
+; UNROLL-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNROLL-NEXT: [[TMP1:%.*]] = getelementptr inbounds [2048 x i32], ptr @b, i64 0, i64 [[INDEX]]
+; UNROLL-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 4
+; UNROLL-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; UNROLL-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; UNROLL-NEXT: [[TMP3:%.*]] = getelementptr inbounds [2048 x i32], ptr @c, i64 0, i64 [[INDEX]]
+; UNROLL-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP3]], i64 4
+; UNROLL-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; UNROLL-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; UNROLL-NEXT: [[TMP5:%.*]] = add nsw <4 x i32> [[WIDE_LOAD2]], [[WIDE_LOAD]]
+; UNROLL-NEXT: [[TMP6:%.*]] = add nsw <4 x i32> [[WIDE_LOAD3]], [[WIDE_LOAD1]]
+; UNROLL-NEXT: [[TMP7:%.*]] = getelementptr inbounds [2048 x i32], ptr @a, i64 0, i64 [[INDEX]]
+; UNROLL-NEXT: [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[TMP7]], i64 4
+; UNROLL-NEXT: store <4 x i32> [[TMP5]], ptr [[TMP7]], align 4
+; UNROLL-NEXT: store <4 x i32> [[TMP6]], ptr [[TMP8]], align 4
+; UNROLL-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; UNROLL-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256
+; UNROLL-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; UNROLL: [[MIDDLE_BLOCK]]:
+; UNROLL-NEXT: br label %[[BB10:.*]]
+; UNROLL: [[BB10]]:
+; UNROLL-NEXT: ret void
+;
br label %1
; <label>:
@@ -45,18 +91,57 @@ define void @example1() {
}
; Select VF=4 because sext <8 x i1> to <8 x i32> is expensive.
-;CHECK-LABEL: @example10b(
-;CHECK: load <4 x i16>
-;CHECK: sext <4 x i16>
-;CHECK: store <4 x i32>
-;CHECK: ret void
-;UNROLL-LABEL: @example10b(
-;UNROLL: load <4 x i16>
-;UNROLL: load <4 x i16>
-;UNROLL: store <4 x i32>
-;UNROLL: store <4 x i32>
-;UNROLL: ret void
define void @example10b(ptr noalias nocapture %sa, ptr noalias nocapture %sb, ptr noalias nocapture %sc, ptr noalias nocapture %ia, ptr noalias nocapture %ib, ptr noalias nocapture %ic) {
+; CHECK-LABEL: define void @example10b(
+; CHECK-SAME: ptr noalias captures(none) [[SA:%.*]], ptr noalias captures(none) [[SB:%.*]], ptr noalias captures(none) [[SC:%.*]], ptr noalias captures(none) [[IA:%.*]], ptr noalias captures(none) [[IB:%.*]], ptr noalias captures(none) [[IC:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i16, ptr [[SB]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i16, ptr [[TMP1]], i64 8
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i16>, ptr [[TMP1]], align 2
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <8 x i16>, ptr [[TMP2]], align 2
+; CHECK-NEXT: [[TMP3:%.*]] = sext <8 x i16> [[WIDE_LOAD]] to <8 x i32>
+; CHECK-NEXT: [[TMP4:%.*]] = sext <8 x i16> [[WIDE_LOAD1]] to <8 x i32>
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[IA]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[TMP5]], i64 8
+; CHECK-NEXT: store <8 x i32> [[TMP3]], ptr [[TMP5]], align 4
+; CHECK-NEXT: store <8 x i32> [[TMP4]], ptr [[TMP6]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[BB8:.*]]
+; CHECK: [[BB8]]:
+; CHECK-NEXT: ret void
+;
+; UNROLL-LABEL: define void @example10b(
+; UNROLL-SAME: ptr noalias captures(none) [[SA:%.*]], ptr noalias captures(none) [[SB:%.*]], ptr noalias captures(none) [[SC:%.*]], ptr noalias captures(none) [[IA:%.*]], ptr noalias captures(none) [[IB:%.*]], ptr noalias captures(none) [[IC:%.*]]) #[[ATTR0]] {
+; UNROLL-NEXT: br label %[[VECTOR_PH:.*]]
+; UNROLL: [[VECTOR_PH]]:
+; UNROLL-NEXT: br label %[[VECTOR_BODY:.*]]
+; UNROLL: [[VECTOR_BODY]]:
+; UNROLL-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNROLL-NEXT: [[TMP1:%.*]] = getelementptr inbounds i16, ptr [[SB]], i64 [[INDEX]]
+; UNROLL-NEXT: [[TMP2:%.*]] = getelementptr inbounds i16, ptr [[TMP1]], i64 8
+; UNROLL-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i16>, ptr [[TMP1]], align 2
+; UNROLL-NEXT: [[WIDE_LOAD1:%.*]] = load <8 x i16>, ptr [[TMP2]], align 2
+; UNROLL-NEXT: [[TMP3:%.*]] = sext <8 x i16> [[WIDE_LOAD]] to <8 x i32>
+; UNROLL-NEXT: [[TMP4:%.*]] = sext <8 x i16> [[WIDE_LOAD1]] to <8 x i32>
+; UNROLL-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[IA]], i64 [[INDEX]]
+; UNROLL-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[TMP5]], i64 8
+; UNROLL-NEXT: store <8 x i32> [[TMP3]], ptr [[TMP5]], align 4
+; UNROLL-NEXT: store <8 x i32> [[TMP4]], ptr [[TMP6]], align 4
+; UNROLL-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; UNROLL-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; UNROLL-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; UNROLL: [[MIDDLE_BLOCK]]:
+; UNROLL-NEXT: br label %[[BB8:.*]]
+; UNROLL: [[BB8]]:
+; UNROLL-NEXT: ret void
+;
br label %1
; <label>:
@@ -75,3 +160,14 @@ define void @example10b(ptr noalias nocapture %sa, ptr noalias nocapture %sb, pt
ret void
}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META1]], [[META2]]}
+;.
+; UNROLL: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; UNROLL: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; UNROLL: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; UNROLL: [[LOOP3]] = distinct !{[[LOOP3]], [[META1]], [[META2]]}
+;.
diff --git a/llvm/test/Transforms/LoopVectorize/X86/induction-costs.ll b/llvm/test/Transforms/LoopVectorize/X86/induction-costs.ll
index afd192246b7ef..f549a5b2b7c69 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/induction-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/induction-costs.ll
@@ -118,23 +118,23 @@ define void @multiple_truncated_ivs_with_wide_uses(i1 %c, ptr %A, ptr %B) {
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i16> [ <i16 0, i16 1, i16 2, i16 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND3:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT6:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[STEP_ADD:%.*]] = add <4 x i16> [[VEC_IND]], splat (i16 4)
-; CHECK-NEXT: [[STEP_ADD4:%.*]] = add <4 x i32> [[VEC_IND3]], splat (i32 4)
-; CHECK-NEXT: [[TMP1:%.*]] = select i1 [[C]], <4 x i16> [[VEC_IND]], <4 x i16> splat (i16 10)
-; CHECK-NEXT: [[TMP2:%.*]] = select i1 [[C]], <4 x i16> [[STEP_ADD]], <4 x i16> splat (i16 10)
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <8 x i16> [ <i16 0, i16 1, i16 2, i16 3, i16 4, i16 5, i16 6, i16 7>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND2:%.*]] = phi <8 x i32> [ <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT4:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[STEP_ADD:%.*]] = add <8 x i16> [[VEC_IND]], splat (i16 8)
+; CHECK-NEXT: [[STEP_ADD3:%.*]] = add <8 x i32> [[VEC_IND2]], splat (i32 8)
+; CHECK-NEXT: [[TMP0:%.*]] = select i1 [[C]], <8 x i16> [[VEC_IND]], <8 x i16> splat (i16 10)
+; CHECK-NEXT: [[TMP1:%.*]] = select i1 [[C]], <8 x i16> [[STEP_ADD]], <8 x i16> splat (i16 10)
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr i16, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i16, ptr [[TMP4]], i64 4
-; CHECK-NEXT: store <4 x i16> [[TMP1]], ptr [[TMP4]], align 2, !alias.scope [[META6:![0-9]+]], !noalias [[META9:![0-9]+]]
-; CHECK-NEXT: store <4 x i16> [[TMP2]], ptr [[TMP3]], align 2, !alias.scope [[META6]], !noalias [[META9]]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i16, ptr [[TMP4]], i64 8
+; CHECK-NEXT: store <8 x i16> [[TMP0]], ptr [[TMP4]], align 2, !alias.scope [[META6:![0-9]+]], !noalias [[META9:![0-9]+]]
+; CHECK-NEXT: store <8 x i16> [[TMP1]], ptr [[TMP3]], align 2, !alias.scope [[META6]], !noalias [[META9]]
; CHECK-NEXT: [[TMP8:%.*]] = getelementptr i32, ptr [[B]], i64 [[INDEX]]
-; CHECK-NEXT: [[TMP5:%.*]] = getelementptr i32, ptr [[TMP8]], i64 4
-; CHECK-NEXT: store <4 x i32> [[VEC_IND3]], ptr [[TMP8]], align 4, !alias.scope [[META9]]
-; CHECK-NEXT: store <4 x i32> [[STEP_ADD4]], ptr [[TMP5]], align 4, !alias.scope [[META9]]
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i16> [[STEP_ADD]], splat (i16 4)
-; CHECK-NEXT: [[VEC_IND_NEXT6]] = add <4 x i32> [[STEP_ADD4]], splat (i32 4)
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr i32, ptr [[TMP8]], i64 8
+; CHECK-NEXT: store <8 x i32> [[VEC_IND2]], ptr [[TMP8]], align 4, !alias.scope [[META9]]
+; CHECK-NEXT: store <8 x i32> [[STEP_ADD3]], ptr [[TMP5]], align 4, !alias.scope [[META9]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add <8 x i16> [[STEP_ADD]], splat (i16 8)
+; CHECK-NEXT: [[VEC_IND_NEXT4]] = add <8 x i32> [[STEP_ADD3]], splat (i32 8)
; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], 64
; CHECK-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
@@ -517,7 +517,7 @@ define i32 @test_scalar_predicated_cost(i64 %x, i64 %y, ptr %A) #0 {
; CHECK-NEXT: [[INDEX_NEXT11]] = add nuw i64 [[INDEX4]], 4
; CHECK-NEXT: [[VEC_IND_NEXT6]] = add <4 x i64> [[VEC_IND5]], splat (i64 4)
; CHECK-NEXT: [[TMP30:%.*]] = icmp eq i64 [[INDEX_NEXT11]], 100
-; CHECK-NEXT: br i1 [[TMP30]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP25:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP30]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[LOOP_HEADER]], !llvm.loop [[LOOP26:![0-9]+]]
; CHECK: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; CHECK-NEXT: br i1 false, label %[[EXIT]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
@@ -536,7 +536,7 @@ define i32 @test_scalar_predicated_cost(i64 %x, i64 %y, ptr %A) #0 {
; CHECK: [[LOOP_LATCH]]:
; CHECK-NEXT: [[IV_NEXT]] = add i64 [[IV]], 1
; CHECK-NEXT: [[EC:%.*]] = icmp eq i64 [[IV]], 100
-; CHECK-NEXT: br i1 [[EC]], label %[[EXIT]], label %[[LOOP_HEADER1]], !llvm.loop [[LOOP26:![0-9]+]]
+; CHECK-NEXT: br i1 [[EC]], label %[[EXIT]], label %[[LOOP_HEADER1]], !llvm.loop [[LOOP27:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret i32 0
;
@@ -616,7 +616,7 @@ define void @wide_iv_trunc(ptr %dst, i64 %N) {
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw <4 x i64> [[VEC_IND]], splat (i64 4)
; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT: br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP27:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP28:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: br label %[[EXIT_LOOPEXIT:.*]]
; CHECK: [[EXIT_LOOPEXIT]]:
@@ -711,11 +711,11 @@ define void @wombat(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; CHECK-NEXT: [[VEC_IND_NEXT]] = add <8 x i32> [[STEP_ADD]], [[TMP2]]
; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], 48
-; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP28:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP29:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: br i1 false, label %[[EXIT:.*]], label %[[VEC_EPILOG_ENTRY:.*]]
; CHECK: [[VEC_EPILOG_ENTRY]]:
-; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF29:![0-9]+]]
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF30:![0-9]+]]
; CHECK: [[VEC_EPILOG_PH]]:
; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ 48, %[[VEC_EPILOG_ENTRY]] ], [ 0, %[[VECTOR_MAIN_LOOP_ENTRY]] ]
; CHECK-NEXT: [[BC_RESUME_VAL3:%.*]] = phi i32 [ [[IND_END]], %[[VEC_EPILOG_ENTRY]] ], [ [[MUL]], %[[VECTOR_MAIN_LOOP_ENTRY]] ]
@@ -741,7 +741,7 @@ define void @wombat(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX11]], 4
; CHECK-NEXT: [[VEC_IND_NEXT14]] = add <4 x i32> [[VEC_IND12]], [[BROADCAST_SPLAT10]]
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT13]], 60
-; CHECK-NEXT: br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP30:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP31:![0-9]+]]
; CHECK: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; CHECK-NEXT: br i1 false, label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
; CHECK: [[VEC_EPILOG_SCALAR_PH]]:
@@ -758,7 +758,7 @@ define void @wombat(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[ADD]] = add i64 [[PHI]], 1
; CHECK-NEXT: [[ICMP:%.*]] = icmp ugt i64 [[PHI]], 65
; CHECK-NEXT: [[TRUNC]] = trunc i64 [[MUL3]] to i32
-; CHECK-NEXT: br i1 [[ICMP]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP31:![0-9]+]]
+; CHECK-NEXT: br i1 [[ICMP]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP32:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -817,11 +817,11 @@ define void @wombat2(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; CHECK-NEXT: [[VEC_IND_NEXT]] = add <8 x i32> [[STEP_ADD]], [[TMP2]]
; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], 48
-; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: br i1 false, label %[[EXIT:.*]], label %[[VEC_EPILOG_ENTRY:.*]]
; CHECK: [[VEC_EPILOG_ENTRY]]:
-; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF29]]
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF30]]
; CHECK: [[VEC_EPILOG_PH]]:
; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ 48, %[[VEC_EPILOG_ENTRY]] ], [ 0, %[[VECTOR_MAIN_LOOP_ENTRY]] ]
; CHECK-NEXT: [[BC_RESUME_VAL3:%.*]] = phi i32 [ [[IND_END]], %[[VEC_EPILOG_ENTRY]] ], [ [[MUL]], %[[VECTOR_MAIN_LOOP_ENTRY]] ]
@@ -847,7 +847,7 @@ define void @wombat2(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX11]], 4
; CHECK-NEXT: [[VEC_IND_NEXT14]] = add <4 x i32> [[VEC_IND12]], [[BROADCAST_SPLAT10]]
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT13]], 60
-; CHECK-NEXT: br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP34:![0-9]+]]
; CHECK: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; CHECK-NEXT: br i1 false, label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
; CHECK: [[VEC_EPILOG_SCALAR_PH]]:
@@ -865,7 +865,7 @@ define void @wombat2(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[ICMP:%.*]] = icmp ugt i64 [[PHI]], 65
; CHECK-NEXT: [[TRUNC_0:%.*]] = trunc i64 [[MUL3]] to i60
; CHECK-NEXT: [[TRUNC_1]] = trunc i60 [[TRUNC_0]] to i32
-; CHECK-NEXT: br i1 [[ICMP]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP34:![0-9]+]]
+; CHECK-NEXT: br i1 [[ICMP]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP35:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -926,11 +926,11 @@ define void @with_dead_use(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; CHECK-NEXT: [[VEC_IND_NEXT]] = add <8 x i32> [[STEP_ADD]], [[TMP2]]
; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], 48
-; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP35:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: br i1 false, label %[[EXIT:.*]], label %[[VEC_EPILOG_ENTRY:.*]]
; CHECK: [[VEC_EPILOG_ENTRY]]:
-; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF29]]
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF30]]
; CHECK: [[VEC_EPILOG_PH]]:
; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ 48, %[[VEC_EPILOG_ENTRY]] ], [ 0, %[[VECTOR_MAIN_LOOP_ENTRY]] ]
; CHECK-NEXT: [[BC_RESUME_VAL3:%.*]] = phi i32 [ [[IND_END]], %[[VEC_EPILOG_ENTRY]] ], [ [[MUL]], %[[VECTOR_MAIN_LOOP_ENTRY]] ]
@@ -956,7 +956,7 @@ define void @with_dead_use(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX11]], 4
; CHECK-NEXT: [[VEC_IND_NEXT14]] = add <4 x i32> [[VEC_IND12]], [[BROADCAST_SPLAT10]]
; CHECK-NEXT: [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT13]], 60
-; CHECK-NEXT: br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP37:![0-9]+]]
; CHECK: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; CHECK-NEXT: br i1 false, label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
; CHECK: [[VEC_EPILOG_SCALAR_PH]]:
@@ -974,7 +974,7 @@ define void @with_dead_use(i32 %arg, ptr %dst) #1 {
; CHECK-NEXT: [[ICMP:%.*]] = icmp ugt i64 [[PHI]], 65
; CHECK-NEXT: [[TRUNC]] = trunc i64 [[MUL3]] to i32
; CHECK-NEXT: [[DEAD_AND:%.*]] = and i32 [[TRUNC]], 123
-; CHECK-NEXT: br i1 [[ICMP]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP37:![0-9]+]]
+; CHECK-NEXT: br i1 [[ICMP]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP38:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
diff --git a/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll b/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll
index 310d894c159ad..806a81f721212 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll
@@ -813,42 +813,15 @@ define void @foo3(ptr nocapture %A, ptr nocapture readonly %B, ptr nocapture rea
; AVX2: [[VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 4
-; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 8
-; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 12
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
-; AVX2-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4, !alias.scope [[META12]]
-; AVX2-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META12]]
-; AVX2-NEXT: [[WIDE_LOAD8:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META12]]
-; AVX2-NEXT: [[TMP4:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD]], splat (i32 100)
-; AVX2-NEXT: [[TMP5:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD6]], splat (i32 100)
-; AVX2-NEXT: [[TMP6:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD7]], splat (i32 100)
-; AVX2-NEXT: [[TMP7:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD8]], splat (i32 100)
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
+; AVX2-NEXT: [[TMP1:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD]], splat (i32 100)
; AVX2-NEXT: [[TMP8:%.*]] = getelementptr double, ptr [[B]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP9:%.*]] = getelementptr double, ptr [[TMP8]], i64 4
-; AVX2-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[TMP8]], i64 8
-; AVX2-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP8]], i64 12
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP8]], <4 x i1> [[TMP4]], <4 x double> poison), !alias.scope [[META15:![0-9]+]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD9:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP9]], <4 x i1> [[TMP5]], <4 x double> poison), !alias.scope [[META15]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD10:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP10]], <4 x i1> [[TMP6]], <4 x double> poison), !alias.scope [[META15]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP11]], <4 x i1> [[TMP7]], <4 x double> poison), !alias.scope [[META15]]
-; AVX2-NEXT: [[TMP12:%.*]] = sitofp <4 x i32> [[WIDE_LOAD]] to <4 x double>
-; AVX2-NEXT: [[TMP13:%.*]] = sitofp <4 x i32> [[WIDE_LOAD6]] to <4 x double>
-; AVX2-NEXT: [[TMP14:%.*]] = sitofp <4 x i32> [[WIDE_LOAD7]] to <4 x double>
-; AVX2-NEXT: [[TMP15:%.*]] = sitofp <4 x i32> [[WIDE_LOAD8]] to <4 x double>
-; AVX2-NEXT: [[TMP16:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD]], [[TMP12]]
-; AVX2-NEXT: [[TMP17:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD9]], [[TMP13]]
-; AVX2-NEXT: [[TMP18:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD10]], [[TMP14]]
-; AVX2-NEXT: [[TMP19:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD11]], [[TMP15]]
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP8]], <8 x i1> [[TMP1]], <8 x double> poison), !alias.scope [[META15:![0-9]+]]
+; AVX2-NEXT: [[TMP3:%.*]] = sitofp <8 x i32> [[WIDE_LOAD]] to <8 x double>
+; AVX2-NEXT: [[TMP4:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], [[TMP3]]
; AVX2-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[A]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP20]], i64 4
-; AVX2-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP20]], i64 8
-; AVX2-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP20]], i64 12
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP16]], ptr align 8 [[TMP20]], <4 x i1> [[TMP4]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP17]], ptr align 8 [[TMP21]], <4 x i1> [[TMP5]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP18]], ptr align 8 [[TMP22]], <4 x i1> [[TMP6]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP19]], ptr align 8 [[TMP23]], <4 x i1> [[TMP7]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP4]], ptr align 8 [[TMP20]], <8 x i1> [[TMP1]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; AVX2-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 10000
; AVX2-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
@@ -878,65 +851,65 @@ define void @foo3(ptr nocapture %A, ptr nocapture readonly %B, ptr nocapture rea
; AVX512: [[VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 8
; AVX512-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 16
-; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 24
-; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
-; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <8 x i32>, ptr [[TMP1]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <8 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[TMP4:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD]], splat (i32 100)
-; AVX512-NEXT: [[TMP5:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD6]], splat (i32 100)
-; AVX512-NEXT: [[TMP6:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD7]], splat (i32 100)
-; AVX512-NEXT: [[TMP7:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD8]], splat (i32 100)
+; AVX512-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 32
+; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 48
+; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
+; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <16 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <16 x i32>, ptr [[TMP9]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <16 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[TMP4:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD]], splat (i32 100)
+; AVX512-NEXT: [[TMP5:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD6]], splat (i32 100)
+; AVX512-NEXT: [[TMP6:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD7]], splat (i32 100)
+; AVX512-NEXT: [[TMP7:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD8]], splat (i32 100)
; AVX512-NEXT: [[TMP8:%.*]] = getelementptr double, ptr [[B]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP9:%.*]] = getelementptr double, ptr [[TMP8]], i64 8
; AVX512-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[TMP8]], i64 16
-; AVX512-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP8]], i64 24
-; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP8]], <8 x i1> [[TMP4]], <8 x double> poison), !alias.scope [[META15:![0-9]+]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD9:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP9]], <8 x i1> [[TMP5]], <8 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD10:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP10]], <8 x i1> [[TMP6]], <8 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP11]], <8 x i1> [[TMP7]], <8 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[TMP12:%.*]] = sitofp <8 x i32> [[WIDE_LOAD]] to <8 x double>
-; AVX512-NEXT: [[TMP13:%.*]] = sitofp <8 x i32> [[WIDE_LOAD6]] to <8 x double>
-; AVX512-NEXT: [[TMP14:%.*]] = sitofp <8 x i32> [[WIDE_LOAD7]] to <8 x double>
-; AVX512-NEXT: [[TMP15:%.*]] = sitofp <8 x i32> [[WIDE_LOAD8]] to <8 x double>
-; AVX512-NEXT: [[TMP16:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], [[TMP12]]
-; AVX512-NEXT: [[TMP17:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD9]], [[TMP13]]
-; AVX512-NEXT: [[TMP18:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD10]], [[TMP14]]
-; AVX512-NEXT: [[TMP19:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD11]], [[TMP15]]
+; AVX512-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP8]], i64 32
+; AVX512-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP8]], i64 48
+; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP8]], <16 x i1> [[TMP4]], <16 x double> poison), !alias.scope [[META15:![0-9]+]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD9:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP10]], <16 x i1> [[TMP5]], <16 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD10:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP21]], <16 x i1> [[TMP6]], <16 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP11]], <16 x i1> [[TMP7]], <16 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[TMP12:%.*]] = sitofp <16 x i32> [[WIDE_LOAD]] to <16 x double>
+; AVX512-NEXT: [[TMP13:%.*]] = sitofp <16 x i32> [[WIDE_LOAD6]] to <16 x double>
+; AVX512-NEXT: [[TMP14:%.*]] = sitofp <16 x i32> [[WIDE_LOAD7]] to <16 x double>
+; AVX512-NEXT: [[TMP15:%.*]] = sitofp <16 x i32> [[WIDE_LOAD8]] to <16 x double>
+; AVX512-NEXT: [[TMP16:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD]], [[TMP12]]
+; AVX512-NEXT: [[TMP17:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD9]], [[TMP13]]
+; AVX512-NEXT: [[TMP18:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD10]], [[TMP14]]
+; AVX512-NEXT: [[TMP19:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD11]], [[TMP15]]
; AVX512-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[A]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP20]], i64 8
; AVX512-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP20]], i64 16
-; AVX512-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP20]], i64 24
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP16]], ptr align 8 [[TMP20]], <8 x i1> [[TMP4]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP17]], ptr align 8 [[TMP21]], <8 x i1> [[TMP5]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP18]], ptr align 8 [[TMP22]], <8 x i1> [[TMP6]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP19]], ptr align 8 [[TMP23]], <8 x i1> [[TMP7]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX512-NEXT: [[TMP32:%.*]] = getelementptr double, ptr [[TMP20]], i64 32
+; AVX512-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP20]], i64 48
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP16]], ptr align 8 [[TMP20]], <16 x i1> [[TMP4]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP17]], ptr align 8 [[TMP22]], <16 x i1> [[TMP5]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP18]], ptr align 8 [[TMP32]], <16 x i1> [[TMP6]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP19]], ptr align 8 [[TMP23]], <16 x i1> [[TMP7]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
; AVX512-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 9984
; AVX512-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: br i1 false, [[FOR_END:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX512: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX512-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF21:![0-9]+]]
+; AVX512-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
; AVX512: [[VEC_EPILOG_PH]]:
; AVX512-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ 9984, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX512-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX512: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX12:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT15:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP25:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[INDEX12]]
-; AVX512-NEXT: [[WIDE_LOAD13:%.*]] = load <8 x i32>, ptr [[TMP25]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[TMP26:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD13]], splat (i32 100)
+; AVX512-NEXT: [[WIDE_LOAD13:%.*]] = load <16 x i32>, ptr [[TMP25]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[TMP26:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD13]], splat (i32 100)
; AVX512-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[B]], i64 [[INDEX12]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD14:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP27]], <8 x i1> [[TMP26]], <8 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[TMP28:%.*]] = sitofp <8 x i32> [[WIDE_LOAD13]] to <8 x double>
-; AVX512-NEXT: [[TMP29:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD14]], [[TMP28]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD14:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP27]], <16 x i1> [[TMP26]], <16 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[TMP28:%.*]] = sitofp <16 x i32> [[WIDE_LOAD13]] to <16 x double>
+; AVX512-NEXT: [[TMP29:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD14]], [[TMP28]]
; AVX512-NEXT: [[TMP30:%.*]] = getelementptr double, ptr [[A]], i64 [[INDEX12]]
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP29]], ptr align 8 [[TMP30]], <8 x i1> [[TMP26]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: [[INDEX_NEXT15]] = add nuw i64 [[INDEX12]], 8
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP29]], ptr align 8 [[TMP30]], <16 x i1> [[TMP26]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: [[INDEX_NEXT15]] = add nuw i64 [[INDEX12]], 16
; AVX512-NEXT: [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT15]], 10000
-; AVX512-NEXT: br i1 [[TMP31]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP22:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP31]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP21:![0-9]+]]
; AVX512: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX512-NEXT: br i1 true, [[FOR_END]], label %[[VEC_EPILOG_SCALAR_PH]]
; AVX512: [[VEC_EPILOG_SCALAR_PH]]:
@@ -1027,21 +1000,21 @@ define void @foo4(ptr nocapture %A, ptr nocapture readonly %B, ptr nocapture rea
; AVX512-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX512: [[VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; AVX512-NEXT: [[VEC_IND:%.*]] = phi <8 x i64> [ <i64 0, i64 16, i64 32, i64 48, i64 64, i64 80, i64 96, i64 112>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; AVX512-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], <8 x i64> [[VEC_IND]]
-; AVX512-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <8 x i32> @llvm.masked.gather.v8i32.v8p0(<8 x ptr> align 4 [[WIDE_GEP]], <8 x i1> splat (i1 true), <8 x i32> poison), !alias.scope [[META24:![0-9]+]]
-; AVX512-NEXT: [[TMP0:%.*]] = icmp slt <8 x i32> [[WIDE_MASKED_GATHER]], splat (i32 100)
-; AVX512-NEXT: [[TMP1:%.*]] = shl nuw nsw <8 x i64> [[VEC_IND]], splat (i64 1)
-; AVX512-NEXT: [[WIDE_GEP6:%.*]] = getelementptr inbounds double, ptr [[B]], <8 x i64> [[TMP1]]
-; AVX512-NEXT: [[WIDE_MASKED_GATHER7:%.*]] = call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 8 [[WIDE_GEP6]], <8 x i1> [[TMP0]], <8 x double> poison), !alias.scope [[META27:![0-9]+]]
-; AVX512-NEXT: [[TMP2:%.*]] = sitofp <8 x i32> [[WIDE_MASKED_GATHER]] to <8 x double>
-; AVX512-NEXT: [[TMP3:%.*]] = fadd <8 x double> [[WIDE_MASKED_GATHER7]], [[TMP2]]
-; AVX512-NEXT: [[WIDE_GEP8:%.*]] = getelementptr inbounds double, ptr [[A]], <8 x i64> [[VEC_IND]]
-; AVX512-NEXT: call void @llvm.masked.scatter.v8f64.v8p0(<8 x double> [[TMP3]], <8 x ptr> align 8 [[WIDE_GEP8]], <8 x i1> [[TMP0]]), !alias.scope [[META29:![0-9]+]], !noalias [[META31:![0-9]+]]
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
-; AVX512-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <8 x i64> [[VEC_IND]], splat (i64 128)
+; AVX512-NEXT: [[VEC_IND:%.*]] = phi <16 x i64> [ <i64 0, i64 16, i64 32, i64 48, i64 64, i64 80, i64 96, i64 112, i64 128, i64 144, i64 160, i64 176, i64 192, i64 208, i64 224, i64 240>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; AVX512-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], <16 x i64> [[VEC_IND]]
+; AVX512-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <16 x i32> @llvm.masked.gather.v16i32.v16p0(<16 x ptr> align 4 [[WIDE_GEP]], <16 x i1> splat (i1 true), <16 x i32> poison), !alias.scope [[META23:![0-9]+]]
+; AVX512-NEXT: [[TMP0:%.*]] = icmp slt <16 x i32> [[WIDE_MASKED_GATHER]], splat (i32 100)
+; AVX512-NEXT: [[TMP1:%.*]] = shl nuw nsw <16 x i64> [[VEC_IND]], splat (i64 1)
+; AVX512-NEXT: [[WIDE_GEP6:%.*]] = getelementptr inbounds double, ptr [[B]], <16 x i64> [[TMP1]]
+; AVX512-NEXT: [[WIDE_MASKED_GATHER7:%.*]] = call <16 x double> @llvm.masked.gather.v16f64.v16p0(<16 x ptr> align 8 [[WIDE_GEP6]], <16 x i1> [[TMP0]], <16 x double> poison), !alias.scope [[META26:![0-9]+]]
+; AVX512-NEXT: [[TMP2:%.*]] = sitofp <16 x i32> [[WIDE_MASKED_GATHER]] to <16 x double>
+; AVX512-NEXT: [[TMP3:%.*]] = fadd <16 x double> [[WIDE_MASKED_GATHER7]], [[TMP2]]
+; AVX512-NEXT: [[WIDE_GEP8:%.*]] = getelementptr inbounds double, ptr [[A]], <16 x i64> [[VEC_IND]]
+; AVX512-NEXT: call void @llvm.masked.scatter.v16f64.v16p0(<16 x double> [[TMP3]], <16 x ptr> align 8 [[WIDE_GEP8]], <16 x i1> [[TMP0]]), !alias.scope [[META28:![0-9]+]], !noalias [[META30:![0-9]+]]
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX512-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <16 x i64> [[VEC_IND]], splat (i64 256)
; AVX512-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], 624
-; AVX512-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP31:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: br label %[[SCALAR_PH]]
; AVX512: [[SCALAR_PH]]:
@@ -1111,49 +1084,19 @@ define void @foo6(ptr nocapture readonly %in, ptr nocapture %out, i32 %size, ptr
; AVX1-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP0:%.*]] = sub i64 4095, [[INDEX]]
; AVX1-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[TMP0]]
-; AVX1-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -3
; AVX1-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -7
-; AVX1-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -11
-; AVX1-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -15
-; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META18:![0-9]+]]
-; AVX1-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META18]]
-; AVX1-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META18]]
-; AVX1-NEXT: [[WIDE_LOAD8:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META18]]
-; AVX1-NEXT: [[REVERSE:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[REVERSE9:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD6]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[REVERSE10:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD7]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[REVERSE11:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD8]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[TMP6:%.*]] = icmp sgt <4 x i32> [[REVERSE]], zeroinitializer
-; AVX1-NEXT: [[TMP7:%.*]] = icmp sgt <4 x i32> [[REVERSE9]], zeroinitializer
-; AVX1-NEXT: [[TMP8:%.*]] = icmp sgt <4 x i32> [[REVERSE10]], zeroinitializer
-; AVX1-NEXT: [[TMP9:%.*]] = icmp sgt <4 x i32> [[REVERSE11]], zeroinitializer
+; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META18:![0-9]+]]
+; AVX1-NEXT: [[REVERSE:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[TMP4:%.*]] = icmp sgt <8 x i32> [[REVERSE]], zeroinitializer
; AVX1-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[IN]], i64 [[TMP0]]
-; AVX1-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP10]], i64 -3
; AVX1-NEXT: [[TMP12:%.*]] = getelementptr double, ptr [[TMP10]], i64 -7
-; AVX1-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -11
-; AVX1-NEXT: [[TMP14:%.*]] = getelementptr double, ptr [[TMP10]], i64 -15
-; AVX1-NEXT: [[REVERSE12:%.*]] = shufflevector <4 x i1> [[TMP6]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[REVERSE13:%.*]] = shufflevector <4 x i1> [[TMP7]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[REVERSE14:%.*]] = shufflevector <4 x i1> [[TMP8]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[REVERSE15:%.*]] = shufflevector <4 x i1> [[TMP9]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP11]], <4 x i1> [[REVERSE12]], <4 x double> poison), !alias.scope [[META21:![0-9]+]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP12]], <4 x i1> [[REVERSE13]], <4 x double> poison), !alias.scope [[META21]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP13]], <4 x i1> [[REVERSE14]], <4 x double> poison), !alias.scope [[META21]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP14]], <4 x i1> [[REVERSE15]], <4 x double> poison), !alias.scope [[META21]]
-; AVX1-NEXT: [[TMP15:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
-; AVX1-NEXT: [[TMP16:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
-; AVX1-NEXT: [[TMP17:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
-; AVX1-NEXT: [[TMP18:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
+; AVX1-NEXT: [[REVERSE6:%.*]] = shufflevector <8 x i1> [[TMP4]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP12]], <8 x i1> [[REVERSE6]], <8 x double> poison), !alias.scope [[META21:![0-9]+]]
+; AVX1-NEXT: [[TMP6:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
; AVX1-NEXT: [[TMP19:%.*]] = getelementptr double, ptr [[OUT]], i64 [[TMP0]]
-; AVX1-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP19]], i64 -3
; AVX1-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP19]], i64 -7
-; AVX1-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -11
-; AVX1-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP19]], i64 -15
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP15]], ptr align 8 [[TMP20]], <4 x i1> [[REVERSE12]]), !alias.scope [[META23:![0-9]+]], !noalias [[META25:![0-9]+]]
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP16]], ptr align 8 [[TMP21]], <4 x i1> [[REVERSE13]]), !alias.scope [[META23]], !noalias [[META25]]
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP17]], ptr align 8 [[TMP22]], <4 x i1> [[REVERSE14]]), !alias.scope [[META23]], !noalias [[META25]]
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP18]], ptr align 8 [[TMP23]], <4 x i1> [[REVERSE15]]), !alias.scope [[META23]], !noalias [[META25]]
-; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX1-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP6]], ptr align 8 [[TMP21]], <8 x i1> [[REVERSE6]]), !alias.scope [[META23:![0-9]+]], !noalias [[META25:![0-9]+]]
+; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; AVX1-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4096
; AVX1-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP26:![0-9]+]]
; AVX1: [[MIDDLE_BLOCK]]:
@@ -1182,49 +1125,19 @@ define void @foo6(ptr nocapture readonly %in, ptr nocapture %out, i32 %size, ptr
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = sub i64 4095, [[INDEX]]
; AVX2-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[TMP0]]
-; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -3
; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -7
-; AVX2-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -11
-; AVX2-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -15
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META22:![0-9]+]]
-; AVX2-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META22]]
-; AVX2-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META22]]
-; AVX2-NEXT: [[WIDE_LOAD8:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META22]]
-; AVX2-NEXT: [[REVERSE:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[REVERSE9:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD6]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[REVERSE10:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD7]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[REVERSE11:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD8]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[TMP6:%.*]] = icmp sgt <4 x i32> [[REVERSE]], zeroinitializer
-; AVX2-NEXT: [[TMP7:%.*]] = icmp sgt <4 x i32> [[REVERSE9]], zeroinitializer
-; AVX2-NEXT: [[TMP8:%.*]] = icmp sgt <4 x i32> [[REVERSE10]], zeroinitializer
-; AVX2-NEXT: [[TMP9:%.*]] = icmp sgt <4 x i32> [[REVERSE11]], zeroinitializer
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META22:![0-9]+]]
+; AVX2-NEXT: [[REVERSE:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[TMP4:%.*]] = icmp sgt <8 x i32> [[REVERSE]], zeroinitializer
; AVX2-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[IN]], i64 [[TMP0]]
-; AVX2-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP10]], i64 -3
; AVX2-NEXT: [[TMP12:%.*]] = getelementptr double, ptr [[TMP10]], i64 -7
-; AVX2-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -11
-; AVX2-NEXT: [[TMP14:%.*]] = getelementptr double, ptr [[TMP10]], i64 -15
-; AVX2-NEXT: [[REVERSE12:%.*]] = shufflevector <4 x i1> [[TMP6]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[REVERSE13:%.*]] = shufflevector <4 x i1> [[TMP7]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[REVERSE14:%.*]] = shufflevector <4 x i1> [[TMP8]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[REVERSE15:%.*]] = shufflevector <4 x i1> [[TMP9]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP11]], <4 x i1> [[REVERSE12]], <4 x double> poison), !alias.scope [[META25:![0-9]+]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP12]], <4 x i1> [[REVERSE13]], <4 x double> poison), !alias.scope [[META25]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP13]], <4 x i1> [[REVERSE14]], <4 x double> poison), !alias.scope [[META25]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP14]], <4 x i1> [[REVERSE15]], <4 x double> poison), !alias.scope [[META25]]
-; AVX2-NEXT: [[TMP15:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
-; AVX2-NEXT: [[TMP16:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
-; AVX2-NEXT: [[TMP17:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
-; AVX2-NEXT: [[TMP18:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
+; AVX2-NEXT: [[REVERSE6:%.*]] = shufflevector <8 x i1> [[TMP4]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP12]], <8 x i1> [[REVERSE6]], <8 x double> poison), !alias.scope [[META25:![0-9]+]]
+; AVX2-NEXT: [[TMP6:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
; AVX2-NEXT: [[TMP19:%.*]] = getelementptr double, ptr [[OUT]], i64 [[TMP0]]
-; AVX2-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP19]], i64 -3
; AVX2-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP19]], i64 -7
-; AVX2-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -11
-; AVX2-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP19]], i64 -15
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP15]], ptr align 8 [[TMP20]], <4 x i1> [[REVERSE12]]), !alias.scope [[META27:![0-9]+]], !noalias [[META29:![0-9]+]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP16]], ptr align 8 [[TMP21]], <4 x i1> [[REVERSE13]]), !alias.scope [[META27]], !noalias [[META29]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP17]], ptr align 8 [[TMP22]], <4 x i1> [[REVERSE14]]), !alias.scope [[META27]], !noalias [[META29]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP18]], ptr align 8 [[TMP23]], <4 x i1> [[REVERSE15]]), !alias.scope [[META27]], !noalias [[META29]]
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP6]], ptr align 8 [[TMP21]], <8 x i1> [[REVERSE6]]), !alias.scope [[META27:![0-9]+]], !noalias [[META29:![0-9]+]]
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; AVX2-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4096
; AVX2-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP30:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
@@ -1253,51 +1166,51 @@ define void @foo6(ptr nocapture readonly %in, ptr nocapture %out, i32 %size, ptr
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP0:%.*]] = sub i64 4095, [[INDEX]]
; AVX512-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[TMP0]]
-; AVX512-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -7
; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -15
-; AVX512-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -23
; AVX512-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -31
-; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META34:![0-9]+]]
-; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META34]]
-; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <8 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META34]]
-; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <8 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META34]]
-; AVX512-NEXT: [[REVERSE:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE9:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD6]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE10:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD7]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE11:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD8]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[TMP6:%.*]] = icmp sgt <8 x i32> [[REVERSE]], zeroinitializer
-; AVX512-NEXT: [[TMP7:%.*]] = icmp sgt <8 x i32> [[REVERSE9]], zeroinitializer
-; AVX512-NEXT: [[TMP8:%.*]] = icmp sgt <8 x i32> [[REVERSE10]], zeroinitializer
-; AVX512-NEXT: [[TMP9:%.*]] = icmp sgt <8 x i32> [[REVERSE11]], zeroinitializer
+; AVX512-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -47
+; AVX512-NEXT: [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -63
+; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META33:![0-9]+]]
+; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <16 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META33]]
+; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <16 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META33]]
+; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <16 x i32>, ptr [[TMP11]], align 4, !alias.scope [[META33]]
+; AVX512-NEXT: [[REVERSE:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE9:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD6]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE10:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD7]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE11:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD8]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[TMP6:%.*]] = icmp sgt <16 x i32> [[REVERSE]], zeroinitializer
+; AVX512-NEXT: [[TMP7:%.*]] = icmp sgt <16 x i32> [[REVERSE9]], zeroinitializer
+; AVX512-NEXT: [[TMP8:%.*]] = icmp sgt <16 x i32> [[REVERSE10]], zeroinitializer
+; AVX512-NEXT: [[TMP9:%.*]] = icmp sgt <16 x i32> [[REVERSE11]], zeroinitializer
; AVX512-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[IN]], i64 [[TMP0]]
-; AVX512-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP10]], i64 -7
; AVX512-NEXT: [[TMP12:%.*]] = getelementptr double, ptr [[TMP10]], i64 -15
-; AVX512-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -23
; AVX512-NEXT: [[TMP14:%.*]] = getelementptr double, ptr [[TMP10]], i64 -31
-; AVX512-NEXT: [[REVERSE12:%.*]] = shufflevector <8 x i1> [[TMP6]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE13:%.*]] = shufflevector <8 x i1> [[TMP7]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE14:%.*]] = shufflevector <8 x i1> [[TMP8]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE15:%.*]] = shufflevector <8 x i1> [[TMP9]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP11]], <8 x i1> [[REVERSE12]], <8 x double> poison), !alias.scope [[META37:![0-9]+]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP12]], <8 x i1> [[REVERSE13]], <8 x double> poison), !alias.scope [[META37]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP13]], <8 x i1> [[REVERSE14]], <8 x double> poison), !alias.scope [[META37]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP14]], <8 x i1> [[REVERSE15]], <8 x double> poison), !alias.scope [[META37]]
-; AVX512-NEXT: [[TMP15:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
-; AVX512-NEXT: [[TMP16:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
-; AVX512-NEXT: [[TMP17:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
-; AVX512-NEXT: [[TMP18:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -47
+; AVX512-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP10]], i64 -63
+; AVX512-NEXT: [[REVERSE12:%.*]] = shufflevector <16 x i1> [[TMP6]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE13:%.*]] = shufflevector <16 x i1> [[TMP7]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE14:%.*]] = shufflevector <16 x i1> [[TMP8]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE15:%.*]] = shufflevector <16 x i1> [[TMP9]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP12]], <16 x i1> [[REVERSE12]], <16 x double> poison), !alias.scope [[META36:![0-9]+]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP14]], <16 x i1> [[REVERSE13]], <16 x double> poison), !alias.scope [[META36]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP13]], <16 x i1> [[REVERSE14]], <16 x double> poison), !alias.scope [[META36]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP20]], <16 x i1> [[REVERSE15]], <16 x double> poison), !alias.scope [[META36]]
+; AVX512-NEXT: [[TMP15:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[TMP16:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[TMP17:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[TMP18:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
; AVX512-NEXT: [[TMP19:%.*]] = getelementptr double, ptr [[OUT]], i64 [[TMP0]]
-; AVX512-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP19]], i64 -7
; AVX512-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP19]], i64 -15
-; AVX512-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -23
; AVX512-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP19]], i64 -31
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP15]], ptr align 8 [[TMP20]], <8 x i1> [[REVERSE12]]), !alias.scope [[META39:![0-9]+]], !noalias [[META41:![0-9]+]]
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP16]], ptr align 8 [[TMP21]], <8 x i1> [[REVERSE13]]), !alias.scope [[META39]], !noalias [[META41]]
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP17]], ptr align 8 [[TMP22]], <8 x i1> [[REVERSE14]]), !alias.scope [[META39]], !noalias [[META41]]
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP18]], ptr align 8 [[TMP23]], <8 x i1> [[REVERSE15]]), !alias.scope [[META39]], !noalias [[META41]]
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX512-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -47
+; AVX512-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP19]], i64 -63
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP15]], ptr align 8 [[TMP21]], <16 x i1> [[REVERSE12]]), !alias.scope [[META38:![0-9]+]], !noalias [[META40:![0-9]+]]
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP16]], ptr align 8 [[TMP23]], <16 x i1> [[REVERSE13]]), !alias.scope [[META38]], !noalias [[META40]]
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP17]], ptr align 8 [[TMP22]], <16 x i1> [[REVERSE14]]), !alias.scope [[META38]], !noalias [[META40]]
+; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP18]], ptr align 8 [[TMP25]], <16 x i1> [[REVERSE15]]), !alias.scope [[META38]], !noalias [[META40]]
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
; AVX512-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4096
-; AVX512-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP42:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP41:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: br [[FOR_END:label %.*]]
; AVX512: [[SCALAR_PH]]:
@@ -1345,84 +1258,54 @@ define void @foo7(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX1-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX1: [[ITER_CHECK]]:
; AVX1-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
+; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX1: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
+; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX1: [[VECTOR_PH]]:
-; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
+; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
; AVX1-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX1-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX1: [[VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX1-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
-; AVX1-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
-; AVX1-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
-; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
-; AVX1-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
-; AVX1-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
-; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
-; AVX1-NEXT: [[TMP4:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX1-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
-; AVX1-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
-; AVX1-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX1-NEXT: [[TMP8:%.*]] = icmp ne <4 x i8> [[TMP4]], zeroinitializer
-; AVX1-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
-; AVX1-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
-; AVX1-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
+; AVX1-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX1-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
; AVX1-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX1-NEXT: [[TMP13:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
-; AVX1-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
-; AVX1-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
-; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP8]], <4 x ptr> poison)
-; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP13]], <4 x i1> [[TMP9]], <4 x ptr> poison)
-; AVX1-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
-; AVX1-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
-; AVX1-NEXT: [[TMP16:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX1-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX1-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
-; AVX1-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
-; AVX1-NEXT: [[TMP20:%.*]] = select <4 x i1> [[TMP8]], <4 x i1> [[TMP16]], <4 x i1> zeroinitializer
-; AVX1-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
-; AVX1-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
-; AVX1-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
+; AVX1-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX1-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
; AVX1-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX1-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
-; AVX1-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
-; AVX1-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP20]])
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP25]], <4 x i1> [[TMP21]])
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
-; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX1-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
+; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; AVX1-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP28:![0-9]+]]
; AVX1: [[MIDDLE_BLOCK]]:
; AVX1-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX1: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
+; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
; AVX1-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF29:![0-9]+]]
; AVX1: [[VEC_EPILOG_PH]]:
; AVX1-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
+; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
; AVX1-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX1-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX1: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_LOAD11:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
-; AVX1-NEXT: [[TMP30:%.*]] = and <4 x i8> [[WIDE_LOAD11]], splat (i8 1)
-; AVX1-NEXT: [[TMP31:%.*]] = icmp ne <4 x i8> [[TMP30]], zeroinitializer
+; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
+; AVX1-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX1-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
; AVX1-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD12:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP31]], <4 x ptr> poison)
-; AVX1-NEXT: [[TMP33:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD12]], splat (ptr null)
-; AVX1-NEXT: [[TMP34:%.*]] = select <4 x i1> [[TMP31]], <4 x i1> [[TMP33]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
+; AVX1-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX1-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
; AVX1-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP34]])
-; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
+; AVX1-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
+; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
; AVX1-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
; AVX1-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP30:![0-9]+]]
; AVX1: [[VEC_EPILOG_MIDDLE_BLOCK]]:
@@ -1437,86 +1320,56 @@ define void @foo7(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX2-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX2: [[ITER_CHECK]]:
; AVX2-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
+; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX2: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
+; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX2: [[VECTOR_PH]]:
-; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
+; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
; AVX2-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX2-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX2: [[VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
-; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
-; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
-; AVX2-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
-; AVX2-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
-; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
-; AVX2-NEXT: [[TMP4:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX2-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
-; AVX2-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
-; AVX2-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX2-NEXT: [[TMP8:%.*]] = icmp ne <4 x i8> [[TMP4]], zeroinitializer
-; AVX2-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
-; AVX2-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
-; AVX2-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
+; AVX2-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX2-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
; AVX2-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP13:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
-; AVX2-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
-; AVX2-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP8]], <4 x ptr> poison)
-; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP13]], <4 x i1> [[TMP9]], <4 x ptr> poison)
-; AVX2-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
-; AVX2-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
-; AVX2-NEXT: [[TMP16:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX2-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX2-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
-; AVX2-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
-; AVX2-NEXT: [[TMP20:%.*]] = select <4 x i1> [[TMP8]], <4 x i1> [[TMP16]], <4 x i1> zeroinitializer
-; AVX2-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
-; AVX2-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
-; AVX2-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
+; AVX2-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX2-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
; AVX2-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
-; AVX2-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
-; AVX2-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP20]])
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP25]], <4 x i1> [[TMP21]])
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX2-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; AVX2-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX2-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX2-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX2: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
-; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF33:![0-9]+]]
+; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
+; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
; AVX2: [[VEC_EPILOG_PH]]:
; AVX2-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
+; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
; AVX2-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX2-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX2: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_LOAD11:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
-; AVX2-NEXT: [[TMP30:%.*]] = and <4 x i8> [[WIDE_LOAD11]], splat (i8 1)
-; AVX2-NEXT: [[TMP31:%.*]] = icmp ne <4 x i8> [[TMP30]], zeroinitializer
+; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
+; AVX2-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX2-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
; AVX2-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD12:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP31]], <4 x ptr> poison)
-; AVX2-NEXT: [[TMP33:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD12]], splat (ptr null)
-; AVX2-NEXT: [[TMP34:%.*]] = select <4 x i1> [[TMP31]], <4 x i1> [[TMP33]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
+; AVX2-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX2-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
; AVX2-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP34]])
-; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
+; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
+; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
; AVX2-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
-; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP34:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
; AVX2: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N14:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC9]]
; AVX2-NEXT: br i1 [[CMP_N14]], [[FOR_END_LOOPEXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
@@ -1532,63 +1385,33 @@ define void @foo7(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX512-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
; AVX512-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX512: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX512-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
+; AVX512-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 64
; AVX512-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX512: [[VECTOR_PH]]:
-; AVX512-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
+; AVX512-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 63
; AVX512-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX512-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX512: [[VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
-; AVX512-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 16
-; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 24
-; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i8>, ptr [[TMP0]], align 1
-; AVX512-NEXT: [[WIDE_LOAD2:%.*]] = load <8 x i8>, ptr [[TMP1]], align 1
-; AVX512-NEXT: [[WIDE_LOAD3:%.*]] = load <8 x i8>, ptr [[TMP2]], align 1
-; AVX512-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP3]], align 1
-; AVX512-NEXT: [[TMP4:%.*]] = and <8 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX512-NEXT: [[TMP5:%.*]] = and <8 x i8> [[WIDE_LOAD2]], splat (i8 1)
-; AVX512-NEXT: [[TMP6:%.*]] = and <8 x i8> [[WIDE_LOAD3]], splat (i8 1)
-; AVX512-NEXT: [[TMP7:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX512-NEXT: [[TMP8:%.*]] = icmp ne <8 x i8> [[TMP4]], zeroinitializer
-; AVX512-NEXT: [[TMP9:%.*]] = icmp ne <8 x i8> [[TMP5]], zeroinitializer
-; AVX512-NEXT: [[TMP10:%.*]] = icmp ne <8 x i8> [[TMP6]], zeroinitializer
-; AVX512-NEXT: [[TMP11:%.*]] = icmp ne <8 x i8> [[TMP7]], zeroinitializer
+; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <64 x i8>, ptr [[TMP0]], align 1
+; AVX512-NEXT: [[TMP2:%.*]] = and <64 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX512-NEXT: [[TMP3:%.*]] = icmp ne <64 x i8> [[TMP2]], zeroinitializer
; AVX512-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP13:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
-; AVX512-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 16
-; AVX512-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 24
-; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP12]], <8 x i1> [[TMP8]], <8 x ptr> poison)
-; AVX512-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP13]], <8 x i1> [[TMP9]], <8 x ptr> poison)
-; AVX512-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP14]], <8 x i1> [[TMP10]], <8 x ptr> poison)
-; AVX512-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP15]], <8 x i1> [[TMP11]], <8 x ptr> poison)
-; AVX512-NEXT: [[TMP16:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX512-NEXT: [[TMP17:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX512-NEXT: [[TMP18:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
-; AVX512-NEXT: [[TMP19:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
-; AVX512-NEXT: [[TMP20:%.*]] = select <8 x i1> [[TMP8]], <8 x i1> [[TMP16]], <8 x i1> zeroinitializer
-; AVX512-NEXT: [[TMP21:%.*]] = select <8 x i1> [[TMP9]], <8 x i1> [[TMP17]], <8 x i1> zeroinitializer
-; AVX512-NEXT: [[TMP22:%.*]] = select <8 x i1> [[TMP10]], <8 x i1> [[TMP18]], <8 x i1> zeroinitializer
-; AVX512-NEXT: [[TMP23:%.*]] = select <8 x i1> [[TMP11]], <8 x i1> [[TMP19]], <8 x i1> zeroinitializer
+; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <64 x ptr> @llvm.masked.load.v64p0.p0(ptr align 8 [[TMP12]], <64 x i1> [[TMP3]], <64 x ptr> poison)
+; AVX512-NEXT: [[TMP5:%.*]] = icmp ne <64 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX512-NEXT: [[TMP6:%.*]] = select <64 x i1> [[TMP3]], <64 x i1> [[TMP5]], <64 x i1> zeroinitializer
; AVX512-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
-; AVX512-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 16
-; AVX512-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 24
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <8 x i1> [[TMP20]])
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP25]], <8 x i1> [[TMP21]])
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <8 x i1> [[TMP22]])
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <8 x i1> [[TMP23]])
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX512-NEXT: call void @llvm.masked.store.v64f64.p0(<64 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <64 x i1> [[TMP6]])
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
; AVX512-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; AVX512-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP44:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP43:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX512-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX512: [[VEC_EPILOG_ITER_CHECK]]:
; AVX512-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
-; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF21]]
+; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF44:![0-9]+]]
; AVX512: [[VEC_EPILOG_PH]]:
; AVX512-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX512-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
@@ -1666,84 +1489,54 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX1-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX1: [[ITER_CHECK]]:
; AVX1-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
+; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX1: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
+; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX1: [[VECTOR_PH]]:
-; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
+; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
; AVX1-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX1-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX1: [[VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX1-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
-; AVX1-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
-; AVX1-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
-; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
-; AVX1-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
-; AVX1-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
-; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
-; AVX1-NEXT: [[TMP4:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX1-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
-; AVX1-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
-; AVX1-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX1-NEXT: [[TMP8:%.*]] = icmp ne <4 x i8> [[TMP4]], zeroinitializer
-; AVX1-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
-; AVX1-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
-; AVX1-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
+; AVX1-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX1-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
; AVX1-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX1-NEXT: [[TMP13:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
-; AVX1-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
-; AVX1-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
-; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP8]], <4 x ptr> poison)
-; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP13]], <4 x i1> [[TMP9]], <4 x ptr> poison)
-; AVX1-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
-; AVX1-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
-; AVX1-NEXT: [[TMP16:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX1-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX1-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
-; AVX1-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
-; AVX1-NEXT: [[TMP20:%.*]] = select <4 x i1> [[TMP8]], <4 x i1> [[TMP16]], <4 x i1> zeroinitializer
-; AVX1-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
-; AVX1-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
-; AVX1-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
+; AVX1-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX1-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
; AVX1-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX1-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
-; AVX1-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
-; AVX1-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP20]])
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP25]], <4 x i1> [[TMP21]])
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
-; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX1-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
+; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; AVX1-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
; AVX1: [[MIDDLE_BLOCK]]:
; AVX1-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX1: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
+; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
; AVX1-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF29]]
; AVX1: [[VEC_EPILOG_PH]]:
; AVX1-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
+; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
; AVX1-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX1-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX1: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_LOAD11:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
-; AVX1-NEXT: [[TMP30:%.*]] = and <4 x i8> [[WIDE_LOAD11]], splat (i8 1)
-; AVX1-NEXT: [[TMP31:%.*]] = icmp ne <4 x i8> [[TMP30]], zeroinitializer
+; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
+; AVX1-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX1-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
; AVX1-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD12:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP31]], <4 x ptr> poison)
-; AVX1-NEXT: [[TMP33:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD12]], splat (ptr null)
-; AVX1-NEXT: [[TMP34:%.*]] = select <4 x i1> [[TMP31]], <4 x i1> [[TMP33]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
+; AVX1-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX1-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
; AVX1-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP34]])
-; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
+; AVX1-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
+; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
; AVX1-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
; AVX1-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
; AVX1: [[VEC_EPILOG_MIDDLE_BLOCK]]:
@@ -1758,86 +1551,56 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX2-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX2: [[ITER_CHECK]]:
; AVX2-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
+; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX2: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
+; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX2: [[VECTOR_PH]]:
-; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
+; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
; AVX2-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX2-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX2: [[VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
-; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
-; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
-; AVX2-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
-; AVX2-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
-; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
-; AVX2-NEXT: [[TMP4:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX2-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
-; AVX2-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
-; AVX2-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX2-NEXT: [[TMP8:%.*]] = icmp ne <4 x i8> [[TMP4]], zeroinitializer
-; AVX2-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
-; AVX2-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
-; AVX2-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
+; AVX2-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX2-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
; AVX2-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP13:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
-; AVX2-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
-; AVX2-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP8]], <4 x ptr> poison)
-; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP13]], <4 x i1> [[TMP9]], <4 x ptr> poison)
-; AVX2-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
-; AVX2-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
-; AVX2-NEXT: [[TMP16:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX2-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX2-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
-; AVX2-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
-; AVX2-NEXT: [[TMP20:%.*]] = select <4 x i1> [[TMP8]], <4 x i1> [[TMP16]], <4 x i1> zeroinitializer
-; AVX2-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
-; AVX2-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
-; AVX2-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
+; AVX2-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX2-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
; AVX2-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX2-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
-; AVX2-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
-; AVX2-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP20]])
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP25]], <4 x i1> [[TMP21]])
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; AVX2-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; AVX2-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; AVX2-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP35:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX2-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX2: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
-; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF33]]
+; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
+; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
; AVX2: [[VEC_EPILOG_PH]]:
; AVX2-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
+; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
; AVX2-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX2-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX2: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_LOAD11:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
-; AVX2-NEXT: [[TMP30:%.*]] = and <4 x i8> [[WIDE_LOAD11]], splat (i8 1)
-; AVX2-NEXT: [[TMP31:%.*]] = icmp ne <4 x i8> [[TMP30]], zeroinitializer
+; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
+; AVX2-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX2-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
; AVX2-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD12:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP31]], <4 x ptr> poison)
-; AVX2-NEXT: [[TMP33:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD12]], splat (ptr null)
-; AVX2-NEXT: [[TMP34:%.*]] = select <4 x i1> [[TMP31]], <4 x i1> [[TMP33]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
+; AVX2-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX2-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
; AVX2-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP34]])
-; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
+; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
+; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
; AVX2-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
-; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP37:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
; AVX2: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N14:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC9]]
; AVX2-NEXT: br i1 [[CMP_N14]], [[FOR_END_LOOPEXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
@@ -1853,55 +1616,25 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX512-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
; AVX512-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX512: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX512-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
+; AVX512-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 64
; AVX512-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX512: [[VECTOR_PH]]:
-; AVX512-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
+; AVX512-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 63
; AVX512-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX512-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX512: [[VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
-; AVX512-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 16
-; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 24
-; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i8>, ptr [[TMP0]], align 1
-; AVX512-NEXT: [[WIDE_LOAD2:%.*]] = load <8 x i8>, ptr [[TMP1]], align 1
-; AVX512-NEXT: [[WIDE_LOAD3:%.*]] = load <8 x i8>, ptr [[TMP2]], align 1
-; AVX512-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP3]], align 1
-; AVX512-NEXT: [[TMP4:%.*]] = and <8 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX512-NEXT: [[TMP5:%.*]] = and <8 x i8> [[WIDE_LOAD2]], splat (i8 1)
-; AVX512-NEXT: [[TMP6:%.*]] = and <8 x i8> [[WIDE_LOAD3]], splat (i8 1)
-; AVX512-NEXT: [[TMP7:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX512-NEXT: [[TMP8:%.*]] = icmp ne <8 x i8> [[TMP4]], zeroinitializer
-; AVX512-NEXT: [[TMP9:%.*]] = icmp ne <8 x i8> [[TMP5]], zeroinitializer
-; AVX512-NEXT: [[TMP10:%.*]] = icmp ne <8 x i8> [[TMP6]], zeroinitializer
-; AVX512-NEXT: [[TMP11:%.*]] = icmp ne <8 x i8> [[TMP7]], zeroinitializer
+; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <64 x i8>, ptr [[TMP0]], align 1
+; AVX512-NEXT: [[TMP2:%.*]] = and <64 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX512-NEXT: [[TMP3:%.*]] = icmp ne <64 x i8> [[TMP2]], zeroinitializer
; AVX512-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP13:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
-; AVX512-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 16
-; AVX512-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 24
-; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP12]], <8 x i1> [[TMP8]], <8 x ptr> poison)
-; AVX512-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP13]], <8 x i1> [[TMP9]], <8 x ptr> poison)
-; AVX512-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP14]], <8 x i1> [[TMP10]], <8 x ptr> poison)
-; AVX512-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP15]], <8 x i1> [[TMP11]], <8 x ptr> poison)
-; AVX512-NEXT: [[TMP16:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX512-NEXT: [[TMP17:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX512-NEXT: [[TMP18:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
-; AVX512-NEXT: [[TMP19:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
-; AVX512-NEXT: [[TMP20:%.*]] = select <8 x i1> [[TMP8]], <8 x i1> [[TMP16]], <8 x i1> zeroinitializer
-; AVX512-NEXT: [[TMP21:%.*]] = select <8 x i1> [[TMP9]], <8 x i1> [[TMP17]], <8 x i1> zeroinitializer
-; AVX512-NEXT: [[TMP22:%.*]] = select <8 x i1> [[TMP10]], <8 x i1> [[TMP18]], <8 x i1> zeroinitializer
-; AVX512-NEXT: [[TMP23:%.*]] = select <8 x i1> [[TMP11]], <8 x i1> [[TMP19]], <8 x i1> zeroinitializer
+; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <64 x ptr> @llvm.masked.load.v64p0.p0(ptr align 8 [[TMP12]], <64 x i1> [[TMP3]], <64 x ptr> poison)
+; AVX512-NEXT: [[TMP5:%.*]] = icmp ne <64 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX512-NEXT: [[TMP6:%.*]] = select <64 x i1> [[TMP3]], <64 x i1> [[TMP5]], <64 x i1> zeroinitializer
; AVX512-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX512-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
-; AVX512-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 16
-; AVX512-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 24
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <8 x i1> [[TMP20]])
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP25]], <8 x i1> [[TMP21]])
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <8 x i1> [[TMP22]])
-; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <8 x i1> [[TMP23]])
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX512-NEXT: call void @llvm.masked.store.v64f64.p0(<64 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <64 x i1> [[TMP6]])
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
; AVX512-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX512-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP47:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
@@ -1909,7 +1642,7 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX512-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX512: [[VEC_EPILOG_ITER_CHECK]]:
; AVX512-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
-; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF21]]
+; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF44]]
; AVX512: [[VEC_EPILOG_PH]]:
; AVX512-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX512-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
@@ -2002,10 +1735,10 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX2: [[VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; AVX2-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP100:%.*]], %[[VECTOR_BODY]] ]
-; AVX2-NEXT: [[VEC_PHI1:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP101:%.*]], %[[VECTOR_BODY]] ]
-; AVX2-NEXT: [[VEC_PHI2:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP102:%.*]], %[[VECTOR_BODY]] ]
-; AVX2-NEXT: [[VEC_PHI3:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP103:%.*]], %[[VECTOR_BODY]] ]
+; AVX2-NEXT: [[VEC_PHI:%.*]] = phi <8 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP196:%.*]], %[[VECTOR_BODY]] ]
+; AVX2-NEXT: [[VEC_PHI1:%.*]] = phi <8 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP197:%.*]], %[[VECTOR_BODY]] ]
+; AVX2-NEXT: [[VEC_PHI2:%.*]] = phi <8 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP198:%.*]], %[[VECTOR_BODY]] ]
+; AVX2-NEXT: [[VEC_PHI3:%.*]] = phi <8 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP199:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = sub i32 100, [[INDEX]]
; AVX2-NEXT: [[TMP1:%.*]] = add i32 [[TMP0]], -1
; AVX2-NEXT: [[TMP2:%.*]] = add i32 [[TMP0]], -2
@@ -2022,6 +1755,22 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: [[TMP13:%.*]] = add i32 [[TMP0]], -13
; AVX2-NEXT: [[TMP14:%.*]] = add i32 [[TMP0]], -14
; AVX2-NEXT: [[TMP15:%.*]] = add i32 [[TMP0]], -15
+; AVX2-NEXT: [[TMP68:%.*]] = add i32 [[TMP0]], -16
+; AVX2-NEXT: [[TMP69:%.*]] = add i32 [[TMP0]], -17
+; AVX2-NEXT: [[TMP70:%.*]] = add i32 [[TMP0]], -18
+; AVX2-NEXT: [[TMP71:%.*]] = add i32 [[TMP0]], -19
+; AVX2-NEXT: [[TMP76:%.*]] = add i32 [[TMP0]], -20
+; AVX2-NEXT: [[TMP77:%.*]] = add i32 [[TMP0]], -21
+; AVX2-NEXT: [[TMP78:%.*]] = add i32 [[TMP0]], -22
+; AVX2-NEXT: [[TMP79:%.*]] = add i32 [[TMP0]], -23
+; AVX2-NEXT: [[TMP96:%.*]] = add i32 [[TMP0]], -24
+; AVX2-NEXT: [[TMP97:%.*]] = add i32 [[TMP0]], -25
+; AVX2-NEXT: [[TMP98:%.*]] = add i32 [[TMP0]], -26
+; AVX2-NEXT: [[TMP99:%.*]] = add i32 [[TMP0]], -27
+; AVX2-NEXT: [[TMP100:%.*]] = add i32 [[TMP0]], -28
+; AVX2-NEXT: [[TMP101:%.*]] = add i32 [[TMP0]], -29
+; AVX2-NEXT: [[TMP102:%.*]] = add i32 [[TMP0]], -30
+; AVX2-NEXT: [[TMP103:%.*]] = add i32 [[TMP0]], -31
; AVX2-NEXT: [[TMP16:%.*]] = zext i32 [[TMP0]] to i64
; AVX2-NEXT: [[TMP17:%.*]] = zext i32 [[TMP1]] to i64
; AVX2-NEXT: [[TMP18:%.*]] = zext i32 [[TMP2]] to i64
@@ -2038,6 +1787,22 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: [[TMP29:%.*]] = zext i32 [[TMP13]] to i64
; AVX2-NEXT: [[TMP30:%.*]] = zext i32 [[TMP14]] to i64
; AVX2-NEXT: [[TMP31:%.*]] = zext i32 [[TMP15]] to i64
+; AVX2-NEXT: [[TMP144:%.*]] = zext i32 [[TMP68]] to i64
+; AVX2-NEXT: [[TMP145:%.*]] = zext i32 [[TMP69]] to i64
+; AVX2-NEXT: [[TMP146:%.*]] = zext i32 [[TMP70]] to i64
+; AVX2-NEXT: [[TMP147:%.*]] = zext i32 [[TMP71]] to i64
+; AVX2-NEXT: [[TMP148:%.*]] = zext i32 [[TMP76]] to i64
+; AVX2-NEXT: [[TMP149:%.*]] = zext i32 [[TMP77]] to i64
+; AVX2-NEXT: [[TMP150:%.*]] = zext i32 [[TMP78]] to i64
+; AVX2-NEXT: [[TMP151:%.*]] = zext i32 [[TMP79]] to i64
+; AVX2-NEXT: [[TMP200:%.*]] = zext i32 [[TMP96]] to i64
+; AVX2-NEXT: [[TMP201:%.*]] = zext i32 [[TMP97]] to i64
+; AVX2-NEXT: [[TMP202:%.*]] = zext i32 [[TMP98]] to i64
+; AVX2-NEXT: [[TMP203:%.*]] = zext i32 [[TMP99]] to i64
+; AVX2-NEXT: [[TMP204:%.*]] = zext i32 [[TMP100]] to i64
+; AVX2-NEXT: [[TMP205:%.*]] = zext i32 [[TMP101]] to i64
+; AVX2-NEXT: [[TMP206:%.*]] = zext i32 [[TMP102]] to i64
+; AVX2-NEXT: [[TMP207:%.*]] = zext i32 [[TMP103]] to i64
; AVX2-NEXT: [[TMP32:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP16]]
; AVX2-NEXT: [[TMP33:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP17]]
; AVX2-NEXT: [[TMP34:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP18]]
@@ -2054,6 +1819,22 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: [[TMP45:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP29]]
; AVX2-NEXT: [[TMP46:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP30]]
; AVX2-NEXT: [[TMP47:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP31]]
+; AVX2-NEXT: [[TMP208:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP144]]
+; AVX2-NEXT: [[TMP209:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP145]]
+; AVX2-NEXT: [[TMP210:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP146]]
+; AVX2-NEXT: [[TMP211:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP147]]
+; AVX2-NEXT: [[TMP84:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP148]]
+; AVX2-NEXT: [[TMP85:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP149]]
+; AVX2-NEXT: [[TMP86:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP150]]
+; AVX2-NEXT: [[TMP87:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP151]]
+; AVX2-NEXT: [[TMP212:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP200]]
+; AVX2-NEXT: [[TMP213:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP201]]
+; AVX2-NEXT: [[TMP214:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP202]]
+; AVX2-NEXT: [[TMP215:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP203]]
+; AVX2-NEXT: [[TMP92:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP204]]
+; AVX2-NEXT: [[TMP93:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP205]]
+; AVX2-NEXT: [[TMP94:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP206]]
+; AVX2-NEXT: [[TMP95:%.*]] = getelementptr [8 x i8], ptr [[P]], i64 [[TMP207]]
; AVX2-NEXT: [[TMP48:%.*]] = load ptr, ptr [[TMP32]], align 8
; AVX2-NEXT: [[TMP49:%.*]] = load ptr, ptr [[TMP33]], align 8
; AVX2-NEXT: [[TMP50:%.*]] = load ptr, ptr [[TMP34]], align 8
@@ -2070,59 +1851,107 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: [[TMP61:%.*]] = load ptr, ptr [[TMP45]], align 8
; AVX2-NEXT: [[TMP62:%.*]] = load ptr, ptr [[TMP46]], align 8
; AVX2-NEXT: [[TMP63:%.*]] = load ptr, ptr [[TMP47]], align 8
+; AVX2-NEXT: [[TMP216:%.*]] = load ptr, ptr [[TMP208]], align 8
+; AVX2-NEXT: [[TMP217:%.*]] = load ptr, ptr [[TMP209]], align 8
+; AVX2-NEXT: [[TMP218:%.*]] = load ptr, ptr [[TMP210]], align 8
+; AVX2-NEXT: [[TMP219:%.*]] = load ptr, ptr [[TMP211]], align 8
+; AVX2-NEXT: [[TMP220:%.*]] = load ptr, ptr [[TMP84]], align 8
+; AVX2-NEXT: [[TMP221:%.*]] = load ptr, ptr [[TMP85]], align 8
+; AVX2-NEXT: [[TMP222:%.*]] = load ptr, ptr [[TMP86]], align 8
+; AVX2-NEXT: [[TMP223:%.*]] = load ptr, ptr [[TMP87]], align 8
+; AVX2-NEXT: [[TMP224:%.*]] = load ptr, ptr [[TMP212]], align 8
+; AVX2-NEXT: [[TMP225:%.*]] = load ptr, ptr [[TMP213]], align 8
+; AVX2-NEXT: [[TMP226:%.*]] = load ptr, ptr [[TMP214]], align 8
+; AVX2-NEXT: [[TMP227:%.*]] = load ptr, ptr [[TMP215]], align 8
+; AVX2-NEXT: [[TMP228:%.*]] = load ptr, ptr [[TMP92]], align 8
+; AVX2-NEXT: [[TMP229:%.*]] = load ptr, ptr [[TMP93]], align 8
+; AVX2-NEXT: [[TMP230:%.*]] = load ptr, ptr [[TMP94]], align 8
+; AVX2-NEXT: [[TMP231:%.*]] = load ptr, ptr [[TMP95]], align 8
; AVX2-NEXT: [[TMP64:%.*]] = load i32, ptr [[TMP48]], align 8
; AVX2-NEXT: [[TMP65:%.*]] = load i32, ptr [[TMP49]], align 8
; AVX2-NEXT: [[TMP66:%.*]] = load i32, ptr [[TMP50]], align 8
; AVX2-NEXT: [[TMP67:%.*]] = load i32, ptr [[TMP51]], align 8
-; AVX2-NEXT: [[TMP68:%.*]] = insertelement <4 x i32> poison, i32 [[TMP64]], i64 0
-; AVX2-NEXT: [[TMP69:%.*]] = insertelement <4 x i32> [[TMP68]], i32 [[TMP65]], i64 1
-; AVX2-NEXT: [[TMP70:%.*]] = insertelement <4 x i32> [[TMP69]], i32 [[TMP66]], i64 2
-; AVX2-NEXT: [[TMP71:%.*]] = insertelement <4 x i32> [[TMP70]], i32 [[TMP67]], i64 3
; AVX2-NEXT: [[TMP72:%.*]] = load i32, ptr [[TMP52]], align 8
; AVX2-NEXT: [[TMP73:%.*]] = load i32, ptr [[TMP53]], align 8
; AVX2-NEXT: [[TMP74:%.*]] = load i32, ptr [[TMP54]], align 8
; AVX2-NEXT: [[TMP75:%.*]] = load i32, ptr [[TMP55]], align 8
-; AVX2-NEXT: [[TMP76:%.*]] = insertelement <4 x i32> poison, i32 [[TMP72]], i64 0
-; AVX2-NEXT: [[TMP77:%.*]] = insertelement <4 x i32> [[TMP76]], i32 [[TMP73]], i64 1
-; AVX2-NEXT: [[TMP78:%.*]] = insertelement <4 x i32> [[TMP77]], i32 [[TMP74]], i64 2
-; AVX2-NEXT: [[TMP79:%.*]] = insertelement <4 x i32> [[TMP78]], i32 [[TMP75]], i64 3
+; AVX2-NEXT: [[TMP232:%.*]] = insertelement <8 x i32> poison, i32 [[TMP64]], i64 0
+; AVX2-NEXT: [[TMP137:%.*]] = insertelement <8 x i32> [[TMP232]], i32 [[TMP65]], i64 1
+; AVX2-NEXT: [[TMP138:%.*]] = insertelement <8 x i32> [[TMP137]], i32 [[TMP66]], i64 2
+; AVX2-NEXT: [[TMP139:%.*]] = insertelement <8 x i32> [[TMP138]], i32 [[TMP67]], i64 3
+; AVX2-NEXT: [[TMP140:%.*]] = insertelement <8 x i32> [[TMP139]], i32 [[TMP72]], i64 4
+; AVX2-NEXT: [[TMP141:%.*]] = insertelement <8 x i32> [[TMP140]], i32 [[TMP73]], i64 5
+; AVX2-NEXT: [[TMP142:%.*]] = insertelement <8 x i32> [[TMP141]], i32 [[TMP74]], i64 6
+; AVX2-NEXT: [[TMP143:%.*]] = insertelement <8 x i32> [[TMP142]], i32 [[TMP75]], i64 7
; AVX2-NEXT: [[TMP80:%.*]] = load i32, ptr [[TMP56]], align 8
; AVX2-NEXT: [[TMP81:%.*]] = load i32, ptr [[TMP57]], align 8
; AVX2-NEXT: [[TMP82:%.*]] = load i32, ptr [[TMP58]], align 8
; AVX2-NEXT: [[TMP83:%.*]] = load i32, ptr [[TMP59]], align 8
-; AVX2-NEXT: [[TMP84:%.*]] = insertelement <4 x i32> poison, i32 [[TMP80]], i64 0
-; AVX2-NEXT: [[TMP85:%.*]] = insertelement <4 x i32> [[TMP84]], i32 [[TMP81]], i64 1
-; AVX2-NEXT: [[TMP86:%.*]] = insertelement <4 x i32> [[TMP85]], i32 [[TMP82]], i64 2
-; AVX2-NEXT: [[TMP87:%.*]] = insertelement <4 x i32> [[TMP86]], i32 [[TMP83]], i64 3
; AVX2-NEXT: [[TMP88:%.*]] = load i32, ptr [[TMP60]], align 8
; AVX2-NEXT: [[TMP89:%.*]] = load i32, ptr [[TMP61]], align 8
; AVX2-NEXT: [[TMP90:%.*]] = load i32, ptr [[TMP62]], align 8
; AVX2-NEXT: [[TMP91:%.*]] = load i32, ptr [[TMP63]], align 8
-; AVX2-NEXT: [[TMP92:%.*]] = insertelement <4 x i32> poison, i32 [[TMP88]], i64 0
-; AVX2-NEXT: [[TMP93:%.*]] = insertelement <4 x i32> [[TMP92]], i32 [[TMP89]], i64 1
-; AVX2-NEXT: [[TMP94:%.*]] = insertelement <4 x i32> [[TMP93]], i32 [[TMP90]], i64 2
-; AVX2-NEXT: [[TMP95:%.*]] = insertelement <4 x i32> [[TMP94]], i32 [[TMP91]], i64 3
-; AVX2-NEXT: [[TMP96:%.*]] = icmp ne <4 x i32> [[TMP71]], zeroinitializer
-; AVX2-NEXT: [[TMP97:%.*]] = icmp ne <4 x i32> [[TMP79]], zeroinitializer
-; AVX2-NEXT: [[TMP98:%.*]] = icmp ne <4 x i32> [[TMP87]], zeroinitializer
-; AVX2-NEXT: [[TMP99:%.*]] = icmp ne <4 x i32> [[TMP95]], zeroinitializer
-; AVX2-NEXT: [[TMP100]] = or <4 x i1> [[VEC_PHI]], [[TMP96]]
-; AVX2-NEXT: [[TMP101]] = or <4 x i1> [[VEC_PHI1]], [[TMP97]]
-; AVX2-NEXT: [[TMP102]] = or <4 x i1> [[VEC_PHI2]], [[TMP98]]
-; AVX2-NEXT: [[TMP103]] = or <4 x i1> [[VEC_PHI3]], [[TMP99]]
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 16
+; AVX2-NEXT: [[TMP152:%.*]] = insertelement <8 x i32> poison, i32 [[TMP80]], i64 0
+; AVX2-NEXT: [[TMP153:%.*]] = insertelement <8 x i32> [[TMP152]], i32 [[TMP81]], i64 1
+; AVX2-NEXT: [[TMP154:%.*]] = insertelement <8 x i32> [[TMP153]], i32 [[TMP82]], i64 2
+; AVX2-NEXT: [[TMP155:%.*]] = insertelement <8 x i32> [[TMP154]], i32 [[TMP83]], i64 3
+; AVX2-NEXT: [[TMP156:%.*]] = insertelement <8 x i32> [[TMP155]], i32 [[TMP88]], i64 4
+; AVX2-NEXT: [[TMP157:%.*]] = insertelement <8 x i32> [[TMP156]], i32 [[TMP89]], i64 5
+; AVX2-NEXT: [[TMP158:%.*]] = insertelement <8 x i32> [[TMP157]], i32 [[TMP90]], i64 6
+; AVX2-NEXT: [[TMP159:%.*]] = insertelement <8 x i32> [[TMP158]], i32 [[TMP91]], i64 7
+; AVX2-NEXT: [[TMP160:%.*]] = load i32, ptr [[TMP216]], align 8
+; AVX2-NEXT: [[TMP161:%.*]] = load i32, ptr [[TMP217]], align 8
+; AVX2-NEXT: [[TMP162:%.*]] = load i32, ptr [[TMP218]], align 8
+; AVX2-NEXT: [[TMP163:%.*]] = load i32, ptr [[TMP219]], align 8
+; AVX2-NEXT: [[TMP164:%.*]] = load i32, ptr [[TMP220]], align 8
+; AVX2-NEXT: [[TMP165:%.*]] = load i32, ptr [[TMP221]], align 8
+; AVX2-NEXT: [[TMP166:%.*]] = load i32, ptr [[TMP222]], align 8
+; AVX2-NEXT: [[TMP167:%.*]] = load i32, ptr [[TMP223]], align 8
+; AVX2-NEXT: [[TMP168:%.*]] = insertelement <8 x i32> poison, i32 [[TMP160]], i64 0
+; AVX2-NEXT: [[TMP169:%.*]] = insertelement <8 x i32> [[TMP168]], i32 [[TMP161]], i64 1
+; AVX2-NEXT: [[TMP170:%.*]] = insertelement <8 x i32> [[TMP169]], i32 [[TMP162]], i64 2
+; AVX2-NEXT: [[TMP171:%.*]] = insertelement <8 x i32> [[TMP170]], i32 [[TMP163]], i64 3
+; AVX2-NEXT: [[TMP172:%.*]] = insertelement <8 x i32> [[TMP171]], i32 [[TMP164]], i64 4
+; AVX2-NEXT: [[TMP173:%.*]] = insertelement <8 x i32> [[TMP172]], i32 [[TMP165]], i64 5
+; AVX2-NEXT: [[TMP174:%.*]] = insertelement <8 x i32> [[TMP173]], i32 [[TMP166]], i64 6
+; AVX2-NEXT: [[TMP175:%.*]] = insertelement <8 x i32> [[TMP174]], i32 [[TMP167]], i64 7
+; AVX2-NEXT: [[TMP176:%.*]] = load i32, ptr [[TMP224]], align 8
+; AVX2-NEXT: [[TMP177:%.*]] = load i32, ptr [[TMP225]], align 8
+; AVX2-NEXT: [[TMP178:%.*]] = load i32, ptr [[TMP226]], align 8
+; AVX2-NEXT: [[TMP179:%.*]] = load i32, ptr [[TMP227]], align 8
+; AVX2-NEXT: [[TMP180:%.*]] = load i32, ptr [[TMP228]], align 8
+; AVX2-NEXT: [[TMP181:%.*]] = load i32, ptr [[TMP229]], align 8
+; AVX2-NEXT: [[TMP182:%.*]] = load i32, ptr [[TMP230]], align 8
+; AVX2-NEXT: [[TMP183:%.*]] = load i32, ptr [[TMP231]], align 8
+; AVX2-NEXT: [[TMP184:%.*]] = insertelement <8 x i32> poison, i32 [[TMP176]], i64 0
+; AVX2-NEXT: [[TMP185:%.*]] = insertelement <8 x i32> [[TMP184]], i32 [[TMP177]], i64 1
+; AVX2-NEXT: [[TMP186:%.*]] = insertelement <8 x i32> [[TMP185]], i32 [[TMP178]], i64 2
+; AVX2-NEXT: [[TMP187:%.*]] = insertelement <8 x i32> [[TMP186]], i32 [[TMP179]], i64 3
+; AVX2-NEXT: [[TMP188:%.*]] = insertelement <8 x i32> [[TMP187]], i32 [[TMP180]], i64 4
+; AVX2-NEXT: [[TMP189:%.*]] = insertelement <8 x i32> [[TMP188]], i32 [[TMP181]], i64 5
+; AVX2-NEXT: [[TMP190:%.*]] = insertelement <8 x i32> [[TMP189]], i32 [[TMP182]], i64 6
+; AVX2-NEXT: [[TMP191:%.*]] = insertelement <8 x i32> [[TMP190]], i32 [[TMP183]], i64 7
+; AVX2-NEXT: [[TMP192:%.*]] = icmp ne <8 x i32> [[TMP143]], zeroinitializer
+; AVX2-NEXT: [[TMP193:%.*]] = icmp ne <8 x i32> [[TMP159]], zeroinitializer
+; AVX2-NEXT: [[TMP194:%.*]] = icmp ne <8 x i32> [[TMP175]], zeroinitializer
+; AVX2-NEXT: [[TMP195:%.*]] = icmp ne <8 x i32> [[TMP191]], zeroinitializer
+; AVX2-NEXT: [[TMP196]] = or <8 x i1> [[VEC_PHI]], [[TMP192]]
+; AVX2-NEXT: [[TMP197]] = or <8 x i1> [[VEC_PHI1]], [[TMP193]]
+; AVX2-NEXT: [[TMP198]] = or <8 x i1> [[VEC_PHI2]], [[TMP194]]
+; AVX2-NEXT: [[TMP199]] = or <8 x i1> [[VEC_PHI3]], [[TMP195]]
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 32
; AVX2-NEXT: [[TMP104:%.*]] = icmp eq i32 [[INDEX_NEXT]], 96
-; AVX2-NEXT: br i1 [[TMP104]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP39:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP104]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP38:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
-; AVX2-NEXT: [[BIN_RDX:%.*]] = or <4 x i1> [[TMP101]], [[TMP100]]
-; AVX2-NEXT: [[BIN_RDX4:%.*]] = or <4 x i1> [[TMP102]], [[BIN_RDX]]
-; AVX2-NEXT: [[BIN_RDX5:%.*]] = or <4 x i1> [[TMP103]], [[BIN_RDX4]]
-; AVX2-NEXT: [[TMP105:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[BIN_RDX5]])
+; AVX2-NEXT: [[BIN_RDX:%.*]] = or <8 x i1> [[TMP197]], [[TMP196]]
+; AVX2-NEXT: [[BIN_RDX4:%.*]] = or <8 x i1> [[TMP198]], [[BIN_RDX]]
+; AVX2-NEXT: [[BIN_RDX5:%.*]] = or <8 x i1> [[TMP199]], [[BIN_RDX4]]
+; AVX2-NEXT: [[TMP105:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[BIN_RDX5]])
; AVX2-NEXT: [[TMP106:%.*]] = freeze i1 [[TMP105]]
; AVX2-NEXT: [[RDX_SELECT:%.*]] = select i1 [[TMP106]], i32 0, i32 1
; AVX2-NEXT: br i1 false, [[EXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX2: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX2-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF33]]
+; AVX2-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF39:![0-9]+]]
; AVX2: [[VEC_EPILOG_PH]]:
; AVX2-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ 96, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX2-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 1, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
diff --git a/llvm/test/Transforms/LoopVectorize/X86/maxbw-cast-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/maxbw-cast-cost.ll
new file mode 100644
index 0000000000000..620cadf0e5012
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/X86/maxbw-cast-cost.ll
@@ -0,0 +1,353 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-vectorize -vectorizer-maximize-bandwidth -force-vector-interleave=1 \
+; RUN: -mcpu=znver4 -S %s | FileCheck %s --check-prefix=MAXBW
+
+target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-linux-gnu"
+
+; i8 x i8 dot product accumulated to i32. Smallest type is i8, widest is
+; i32. With MaxBW enabled, the VF is chosen from the smallest type, so
+; MaxBW should choose VF=64 (512/8) rather than the natural VF=16 (512/32).
+define void @dot_i8(ptr noalias %weights, ptr noalias %input, ptr noalias %output, i64 %n) {
+; MAXBW-LABEL: define void @dot_i8(
+; MAXBW-SAME: ptr noalias [[WEIGHTS:%.*]], ptr noalias [[INPUT:%.*]], ptr noalias [[OUTPUT:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; MAXBW-NEXT: [[ITER_CHECK:.*]]:
+; MAXBW-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; MAXBW-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; MAXBW: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; MAXBW-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 64
+; MAXBW-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; MAXBW: [[VECTOR_PH]]:
+; MAXBW-NEXT: [[N_MOD_VF:%.*]] = and i64 [[N]], 63
+; MAXBW-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; MAXBW-NEXT: br label %[[VECTOR_BODY:.*]]
+; MAXBW: [[VECTOR_BODY]]:
+; MAXBW-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; MAXBW-NEXT: [[VEC_PHI:%.*]] = phi <64 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP7:%.*]], %[[VECTOR_BODY]] ]
+; MAXBW-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[WEIGHTS]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD:%.*]] = load <64 x i8>, ptr [[TMP0]], align 1
+; MAXBW-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[INPUT]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD2:%.*]] = load <64 x i8>, ptr [[TMP1]], align 1
+; MAXBW-NEXT: [[TMP2:%.*]] = sext <64 x i8> [[WIDE_LOAD]] to <64 x i32>
+; MAXBW-NEXT: [[TMP3:%.*]] = zext <64 x i8> [[WIDE_LOAD2]] to <64 x i32>
+; MAXBW-NEXT: [[TMP4:%.*]] = mul nsw <64 x i32> [[TMP2]], [[TMP3]]
+; MAXBW-NEXT: [[TMP7]] = add <64 x i32> [[TMP4]], [[VEC_PHI]]
+; MAXBW-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
+; MAXBW-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; MAXBW-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; MAXBW: [[MIDDLE_BLOCK]]:
+; MAXBW-NEXT: [[TMP8:%.*]] = call i32 @llvm.vector.reduce.add.v64i32(<64 x i32> [[TMP7]])
+; MAXBW-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; MAXBW-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; MAXBW: [[VEC_EPILOG_ITER_CHECK]]:
+; MAXBW-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
+; MAXBW-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3:![0-9]+]]
+; MAXBW: [[VEC_EPILOG_PH]]:
+; MAXBW-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; MAXBW-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP8]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; MAXBW-NEXT: [[N_MOD_VF3:%.*]] = and i64 [[N]], 7
+; MAXBW-NEXT: [[N_VEC4:%.*]] = sub i64 [[N]], [[N_MOD_VF3]]
+; MAXBW-NEXT: [[TMP17:%.*]] = insertelement <8 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
+; MAXBW-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; MAXBW: [[VEC_EPILOG_VECTOR_BODY]]:
+; MAXBW-NEXT: [[INDEX5:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT9:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; MAXBW-NEXT: [[VEC_PHI6:%.*]] = phi <8 x i32> [ [[TMP17]], %[[VEC_EPILOG_PH]] ], [ [[TMP14:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; MAXBW-NEXT: [[TMP9:%.*]] = getelementptr inbounds i8, ptr [[WEIGHTS]], i64 [[INDEX5]]
+; MAXBW-NEXT: [[WIDE_LOAD7:%.*]] = load <8 x i8>, ptr [[TMP9]], align 1
+; MAXBW-NEXT: [[TMP10:%.*]] = getelementptr inbounds i8, ptr [[INPUT]], i64 [[INDEX5]]
+; MAXBW-NEXT: [[WIDE_LOAD8:%.*]] = load <8 x i8>, ptr [[TMP10]], align 1
+; MAXBW-NEXT: [[TMP11:%.*]] = sext <8 x i8> [[WIDE_LOAD7]] to <8 x i32>
+; MAXBW-NEXT: [[TMP12:%.*]] = zext <8 x i8> [[WIDE_LOAD8]] to <8 x i32>
+; MAXBW-NEXT: [[TMP13:%.*]] = mul nsw <8 x i32> [[TMP11]], [[TMP12]]
+; MAXBW-NEXT: [[TMP14]] = add <8 x i32> [[TMP13]], [[VEC_PHI6]]
+; MAXBW-NEXT: [[INDEX_NEXT9]] = add nuw i64 [[INDEX5]], 8
+; MAXBW-NEXT: [[TMP15:%.*]] = icmp eq i64 [[INDEX_NEXT9]], [[N_VEC4]]
+; MAXBW-NEXT: br i1 [[TMP15]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; MAXBW: [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; MAXBW-NEXT: [[TMP16:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP14]])
+; MAXBW-NEXT: [[CMP_N10:%.*]] = icmp eq i64 [[N]], [[N_VEC4]]
+; MAXBW-NEXT: br i1 [[CMP_N10]], label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; MAXBW: [[VEC_EPILOG_SCALAR_PH]]:
+; MAXBW-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC4]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; MAXBW-NEXT: [[BC_MERGE_RDX10:%.*]] = phi i32 [ [[TMP16]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP8]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; MAXBW-NEXT: br label %[[LOOP:.*]]
+; MAXBW: [[LOOP]]:
+; MAXBW-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; MAXBW-NEXT: [[SUM:%.*]] = phi i32 [ [[BC_MERGE_RDX10]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[ADD:%.*]], %[[LOOP]] ]
+; MAXBW-NEXT: [[GW:%.*]] = getelementptr inbounds i8, ptr [[WEIGHTS]], i64 [[IV]]
+; MAXBW-NEXT: [[W:%.*]] = load i8, ptr [[GW]], align 1
+; MAXBW-NEXT: [[GI:%.*]] = getelementptr inbounds i8, ptr [[INPUT]], i64 [[IV]]
+; MAXBW-NEXT: [[I:%.*]] = load i8, ptr [[GI]], align 1
+; MAXBW-NEXT: [[WE:%.*]] = sext i8 [[W]] to i32
+; MAXBW-NEXT: [[IE:%.*]] = zext i8 [[I]] to i32
+; MAXBW-NEXT: [[MUL:%.*]] = mul nsw i32 [[WE]], [[IE]]
+; MAXBW-NEXT: [[ADD]] = add nsw i32 [[MUL]], [[SUM]]
+; MAXBW-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; MAXBW-NEXT: [[COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; MAXBW-NEXT: br i1 [[COND]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; MAXBW: [[EXIT]]:
+; MAXBW-NEXT: [[ADD_LCSSA:%.*]] = phi i32 [ [[ADD]], %[[LOOP]] ], [ [[TMP8]], %[[MIDDLE_BLOCK]] ], [ [[TMP16]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; MAXBW-NEXT: store i32 [[ADD_LCSSA]], ptr [[OUTPUT]], align 4
+; MAXBW-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %sum = phi i32 [ 0, %entry ], [ %add, %loop ]
+ %gw = getelementptr inbounds i8, ptr %weights, i64 %iv
+ %w = load i8, ptr %gw
+ %gi = getelementptr inbounds i8, ptr %input, i64 %iv
+ %i = load i8, ptr %gi
+ %we = sext i8 %w to i32
+ %ie = zext i8 %i to i32
+ %mul = mul nsw i32 %we, %ie
+ %add = add nsw i32 %mul, %sum
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cond = icmp eq i64 %iv.next, %n
+ br i1 %cond, label %exit, label %loop
+
+exit:
+ store i32 %add, ptr %output
+ ret void
+}
+
+; Four i8 loads, sext to i32, multiply pairs, add, store i32.
+; Smallest type=i8, widest=i32. MaxBW widens to VF=64.
+define void @multi_sext_dot(ptr noalias %a, ptr noalias %b,
+; MAXBW-LABEL: define void @multi_sext_dot(
+; MAXBW-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]], ptr noalias [[D:%.*]], ptr noalias [[OUT:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; MAXBW-NEXT: [[ITER_CHECK:.*]]:
+; MAXBW-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; MAXBW-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; MAXBW: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; MAXBW-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 64
+; MAXBW-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; MAXBW: [[VECTOR_PH]]:
+; MAXBW-NEXT: [[N_MOD_VF:%.*]] = and i64 [[N]], 63
+; MAXBW-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; MAXBW-NEXT: br label %[[VECTOR_BODY:.*]]
+; MAXBW: [[VECTOR_BODY]]:
+; MAXBW-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; MAXBW-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD:%.*]] = load <64 x i8>, ptr [[TMP0]], align 1
+; MAXBW-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD2:%.*]] = load <64 x i8>, ptr [[TMP1]], align 1
+; MAXBW-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[C]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD3:%.*]] = load <64 x i8>, ptr [[TMP2]], align 1
+; MAXBW-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[D]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD4:%.*]] = load <64 x i8>, ptr [[TMP3]], align 1
+; MAXBW-NEXT: [[TMP4:%.*]] = sext <64 x i8> [[WIDE_LOAD]] to <64 x i32>
+; MAXBW-NEXT: [[TMP5:%.*]] = sext <64 x i8> [[WIDE_LOAD2]] to <64 x i32>
+; MAXBW-NEXT: [[TMP6:%.*]] = sext <64 x i8> [[WIDE_LOAD3]] to <64 x i32>
+; MAXBW-NEXT: [[TMP7:%.*]] = sext <64 x i8> [[WIDE_LOAD4]] to <64 x i32>
+; MAXBW-NEXT: [[TMP8:%.*]] = mul nsw <64 x i32> [[TMP4]], [[TMP5]]
+; MAXBW-NEXT: [[TMP9:%.*]] = mul nsw <64 x i32> [[TMP6]], [[TMP7]]
+; MAXBW-NEXT: [[TMP10:%.*]] = add nsw <64 x i32> [[TMP8]], [[TMP9]]
+; MAXBW-NEXT: [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[OUT]], i64 [[INDEX]]
+; MAXBW-NEXT: store <64 x i32> [[TMP10]], ptr [[TMP11]], align 4
+; MAXBW-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
+; MAXBW-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; MAXBW-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; MAXBW: [[MIDDLE_BLOCK]]:
+; MAXBW-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; MAXBW-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; MAXBW: [[VEC_EPILOG_ITER_CHECK]]:
+; MAXBW-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
+; MAXBW-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
+; MAXBW: [[VEC_EPILOG_PH]]:
+; MAXBW-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; MAXBW-NEXT: [[N_MOD_VF5:%.*]] = and i64 [[N]], 7
+; MAXBW-NEXT: [[N_VEC6:%.*]] = sub i64 [[N]], [[N_MOD_VF5]]
+; MAXBW-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; MAXBW: [[VEC_EPILOG_VECTOR_BODY]]:
+; MAXBW-NEXT: [[INDEX7:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT12:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; MAXBW-NEXT: [[TMP13:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX7]]
+; MAXBW-NEXT: [[WIDE_LOAD8:%.*]] = load <8 x i8>, ptr [[TMP13]], align 1
+; MAXBW-NEXT: [[TMP14:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[INDEX7]]
+; MAXBW-NEXT: [[WIDE_LOAD9:%.*]] = load <8 x i8>, ptr [[TMP14]], align 1
+; MAXBW-NEXT: [[TMP15:%.*]] = getelementptr inbounds i8, ptr [[C]], i64 [[INDEX7]]
+; MAXBW-NEXT: [[WIDE_LOAD10:%.*]] = load <8 x i8>, ptr [[TMP15]], align 1
+; MAXBW-NEXT: [[TMP16:%.*]] = getelementptr inbounds i8, ptr [[D]], i64 [[INDEX7]]
+; MAXBW-NEXT: [[WIDE_LOAD11:%.*]] = load <8 x i8>, ptr [[TMP16]], align 1
+; MAXBW-NEXT: [[TMP17:%.*]] = sext <8 x i8> [[WIDE_LOAD8]] to <8 x i32>
+; MAXBW-NEXT: [[TMP18:%.*]] = sext <8 x i8> [[WIDE_LOAD9]] to <8 x i32>
+; MAXBW-NEXT: [[TMP19:%.*]] = sext <8 x i8> [[WIDE_LOAD10]] to <8 x i32>
+; MAXBW-NEXT: [[TMP20:%.*]] = sext <8 x i8> [[WIDE_LOAD11]] to <8 x i32>
+; MAXBW-NEXT: [[TMP21:%.*]] = mul nsw <8 x i32> [[TMP17]], [[TMP18]]
+; MAXBW-NEXT: [[TMP22:%.*]] = mul nsw <8 x i32> [[TMP19]], [[TMP20]]
+; MAXBW-NEXT: [[TMP23:%.*]] = add nsw <8 x i32> [[TMP21]], [[TMP22]]
+; MAXBW-NEXT: [[TMP24:%.*]] = getelementptr inbounds i32, ptr [[OUT]], i64 [[INDEX7]]
+; MAXBW-NEXT: store <8 x i32> [[TMP23]], ptr [[TMP24]], align 4
+; MAXBW-NEXT: [[INDEX_NEXT12]] = add nuw i64 [[INDEX7]], 8
+; MAXBW-NEXT: [[TMP25:%.*]] = icmp eq i64 [[INDEX_NEXT12]], [[N_VEC6]]
+; MAXBW-NEXT: br i1 [[TMP25]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; MAXBW: [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; MAXBW-NEXT: [[CMP_N13:%.*]] = icmp eq i64 [[N]], [[N_VEC6]]
+; MAXBW-NEXT: br i1 [[CMP_N13]], label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; MAXBW: [[VEC_EPILOG_SCALAR_PH]]:
+; MAXBW-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC6]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; MAXBW-NEXT: br label %[[LOOP:.*]]
+; MAXBW: [[LOOP]]:
+; MAXBW-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; MAXBW-NEXT: [[GA:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV]]
+; MAXBW-NEXT: [[LA:%.*]] = load i8, ptr [[GA]], align 1
+; MAXBW-NEXT: [[GB:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV]]
+; MAXBW-NEXT: [[LB:%.*]] = load i8, ptr [[GB]], align 1
+; MAXBW-NEXT: [[GC:%.*]] = getelementptr inbounds i8, ptr [[C]], i64 [[IV]]
+; MAXBW-NEXT: [[LC:%.*]] = load i8, ptr [[GC]], align 1
+; MAXBW-NEXT: [[GD:%.*]] = getelementptr inbounds i8, ptr [[D]], i64 [[IV]]
+; MAXBW-NEXT: [[LD:%.*]] = load i8, ptr [[GD]], align 1
+; MAXBW-NEXT: [[EA:%.*]] = sext i8 [[LA]] to i32
+; MAXBW-NEXT: [[EB:%.*]] = sext i8 [[LB]] to i32
+; MAXBW-NEXT: [[EC:%.*]] = sext i8 [[LC]] to i32
+; MAXBW-NEXT: [[ED:%.*]] = sext i8 [[LD]] to i32
+; MAXBW-NEXT: [[M1:%.*]] = mul nsw i32 [[EA]], [[EB]]
+; MAXBW-NEXT: [[M2:%.*]] = mul nsw i32 [[EC]], [[ED]]
+; MAXBW-NEXT: [[SUM:%.*]] = add nsw i32 [[M1]], [[M2]]
+; MAXBW-NEXT: [[GO:%.*]] = getelementptr inbounds i32, ptr [[OUT]], i64 [[IV]]
+; MAXBW-NEXT: store i32 [[SUM]], ptr [[GO]], align 4
+; MAXBW-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; MAXBW-NEXT: [[COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; MAXBW-NEXT: br i1 [[COND]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
+; MAXBW: [[EXIT]]:
+; MAXBW-NEXT: ret void
+;
+ ptr noalias %c, ptr noalias %d,
+ ptr noalias %out, i64 %n) {
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %ga = getelementptr inbounds i8, ptr %a, i64 %iv
+ %la = load i8, ptr %ga
+ %gb = getelementptr inbounds i8, ptr %b, i64 %iv
+ %lb = load i8, ptr %gb
+ %gc = getelementptr inbounds i8, ptr %c, i64 %iv
+ %lc = load i8, ptr %gc
+ %gd = getelementptr inbounds i8, ptr %d, i64 %iv
+ %ld = load i8, ptr %gd
+ %ea = sext i8 %la to i32
+ %eb = sext i8 %lb to i32
+ %ec = sext i8 %lc to i32
+ %ed = sext i8 %ld to i32
+ %m1 = mul nsw i32 %ea, %eb
+ %m2 = mul nsw i32 %ec, %ed
+ %sum = add nsw i32 %m1, %m2
+ %go = getelementptr inbounds i32, ptr %out, i64 %iv
+ store i32 %sum, ptr %go
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cond = icmp eq i64 %iv.next, %n
+ br i1 %cond, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+; i16 add — smallest = widest = i16. Both MaxBW and default select
+; VF = 512/16 = 32. Control case verifying no regression.
+define void @add_i16(ptr noalias %a, ptr noalias %b, ptr noalias %c, i64 %n) {
+; MAXBW-LABEL: define void @add_i16(
+; MAXBW-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; MAXBW-NEXT: [[ITER_CHECK:.*]]:
+; MAXBW-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; MAXBW-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; MAXBW: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; MAXBW-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; MAXBW-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; MAXBW: [[VECTOR_PH]]:
+; MAXBW-NEXT: [[N_MOD_VF:%.*]] = and i64 [[N]], 31
+; MAXBW-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; MAXBW-NEXT: br label %[[VECTOR_BODY:.*]]
+; MAXBW: [[VECTOR_BODY]]:
+; MAXBW-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; MAXBW-NEXT: [[TMP0:%.*]] = getelementptr inbounds i16, ptr [[A]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i16>, ptr [[TMP0]], align 2
+; MAXBW-NEXT: [[TMP1:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[INDEX]]
+; MAXBW-NEXT: [[WIDE_LOAD2:%.*]] = load <32 x i16>, ptr [[TMP1]], align 2
+; MAXBW-NEXT: [[TMP2:%.*]] = add <32 x i16> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; MAXBW-NEXT: [[TMP3:%.*]] = getelementptr inbounds i16, ptr [[C]], i64 [[INDEX]]
+; MAXBW-NEXT: store <32 x i16> [[TMP2]], ptr [[TMP3]], align 2
+; MAXBW-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; MAXBW-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; MAXBW-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
+; MAXBW: [[MIDDLE_BLOCK]]:
+; MAXBW-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; MAXBW-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; MAXBW: [[VEC_EPILOG_ITER_CHECK]]:
+; MAXBW-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
+; MAXBW-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF10:![0-9]+]]
+; MAXBW: [[VEC_EPILOG_PH]]:
+; MAXBW-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; MAXBW-NEXT: [[N_MOD_VF3:%.*]] = and i64 [[N]], 3
+; MAXBW-NEXT: [[N_VEC4:%.*]] = sub i64 [[N]], [[N_MOD_VF3]]
+; MAXBW-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; MAXBW: [[VEC_EPILOG_VECTOR_BODY]]:
+; MAXBW-NEXT: [[INDEX5:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT8:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; MAXBW-NEXT: [[TMP5:%.*]] = getelementptr inbounds i16, ptr [[A]], i64 [[INDEX5]]
+; MAXBW-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i16>, ptr [[TMP5]], align 2
+; MAXBW-NEXT: [[TMP6:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[INDEX5]]
+; MAXBW-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i16>, ptr [[TMP6]], align 2
+; MAXBW-NEXT: [[TMP7:%.*]] = add <4 x i16> [[WIDE_LOAD6]], [[WIDE_LOAD7]]
+; MAXBW-NEXT: [[TMP8:%.*]] = getelementptr inbounds i16, ptr [[C]], i64 [[INDEX5]]
+; MAXBW-NEXT: store <4 x i16> [[TMP7]], ptr [[TMP8]], align 2
+; MAXBW-NEXT: [[INDEX_NEXT8]] = add nuw i64 [[INDEX5]], 4
+; MAXBW-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT8]], [[N_VEC4]]
+; MAXBW-NEXT: br i1 [[TMP9]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; MAXBW: [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; MAXBW-NEXT: [[CMP_N9:%.*]] = icmp eq i64 [[N]], [[N_VEC4]]
+; MAXBW-NEXT: br i1 [[CMP_N9]], label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; MAXBW: [[VEC_EPILOG_SCALAR_PH]]:
+; MAXBW-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC4]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; MAXBW-NEXT: br label %[[LOOP:.*]]
+; MAXBW: [[LOOP]]:
+; MAXBW-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; MAXBW-NEXT: [[GA:%.*]] = getelementptr inbounds i16, ptr [[A]], i64 [[IV]]
+; MAXBW-NEXT: [[LA:%.*]] = load i16, ptr [[GA]], align 2
+; MAXBW-NEXT: [[GB:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[IV]]
+; MAXBW-NEXT: [[LB:%.*]] = load i16, ptr [[GB]], align 2
+; MAXBW-NEXT: [[ADD:%.*]] = add i16 [[LA]], [[LB]]
+; MAXBW-NEXT: [[GC:%.*]] = getelementptr inbounds i16, ptr [[C]], i64 [[IV]]
+; MAXBW-NEXT: store i16 [[ADD]], ptr [[GC]], align 2
+; MAXBW-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; MAXBW-NEXT: [[COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; MAXBW-NEXT: br i1 [[COND]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP12:![0-9]+]]
+; MAXBW: [[EXIT]]:
+; MAXBW-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %ga = getelementptr inbounds i16, ptr %a, i64 %iv
+ %la = load i16, ptr %ga
+ %gb = getelementptr inbounds i16, ptr %b, i64 %iv
+ %lb = load i16, ptr %gb
+ %add = add i16 %la, %lb
+ %gc = getelementptr inbounds i16, ptr %c, i64 %iv
+ store i16 %add, ptr %gc
+ %iv.next = add nuw nsw i64 %iv, 1
+ %cond = icmp eq i64 %iv.next, %n
+ br i1 %cond, label %exit, label %loop
+
+exit:
+ ret void
+}
+;.
+; MAXBW: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; MAXBW: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; MAXBW: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; MAXBW: [[PROF3]] = !{!"branch_weights", i32 8, i32 56}
+; MAXBW: [[LOOP4]] = distinct !{[[LOOP4]], [[META1]], [[META2]]}
+; MAXBW: [[LOOP5]] = distinct !{[[LOOP5]], [[META2]], [[META1]]}
+; MAXBW: [[LOOP6]] = distinct !{[[LOOP6]], [[META1]], [[META2]]}
+; MAXBW: [[LOOP7]] = distinct !{[[LOOP7]], [[META1]], [[META2]]}
+; MAXBW: [[LOOP8]] = distinct !{[[LOOP8]], [[META2]], [[META1]]}
+; MAXBW: [[LOOP9]] = distinct !{[[LOOP9]], [[META1]], [[META2]]}
+; MAXBW: [[PROF10]] = !{!"branch_weights", i32 4, i32 28}
+; MAXBW: [[LOOP11]] = distinct !{[[LOOP11]], [[META1]], [[META2]]}
+; MAXBW: [[LOOP12]] = distinct !{[[LOOP12]], [[META2]], [[META1]]}
+;.
diff --git a/llvm/test/Transforms/LoopVectorize/X86/no_fpmath.ll b/llvm/test/Transforms/LoopVectorize/X86/no_fpmath.ll
index ae77b4270ab3e..2056547c2c988 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/no_fpmath.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/no_fpmath.ll
@@ -2,7 +2,7 @@
; CHECK: remark: no_fpmath.c:6:11: loop not vectorized: cannot prove it is safe to reorder floating-point operations
; CHECK: remark: no_fpmath.c:6:14: loop not vectorized
-; CHECK: remark: no_fpmath.c:17:14: vectorized loop (vectorization width: 2, interleaved count: 2)
+; CHECK: remark: no_fpmath.c:17:14: vectorized loop (vectorization width: 4, interleaved count: 2)
target datalayout = "e-m:o-i64:64-f80:128-n8:16:32:64-S128"
target triple = "x86_64-apple-macosx10.10.0"
diff --git a/llvm/test/Transforms/LoopVectorize/X86/no_fpmath_with_hotness.ll b/llvm/test/Transforms/LoopVectorize/X86/no_fpmath_with_hotness.ll
index cbedbf7fd10c0..6406756f1eead 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/no_fpmath_with_hotness.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/no_fpmath_with_hotness.ll
@@ -2,7 +2,7 @@
; CHECK: remark: no_fpmath.c:6:11: loop not vectorized: cannot prove it is safe to reorder floating-point operations (hotness: 300)
; CHECK: remark: no_fpmath.c:6:14: loop not vectorized
-; CHECK: remark: no_fpmath.c:17:14: vectorized loop (vectorization width: 2, interleaved count: 1) (hotness: 300)
+; CHECK: remark: no_fpmath.c:17:14: vectorized loop (vectorization width: 4, interleaved count: 1) (hotness: 300)
target datalayout = "e-m:o-i64:64-f80:128-n8:16:32:64-S128"
target triple = "x86_64-apple-macosx10.10.0"
diff --git a/llvm/test/Transforms/LoopVectorize/X86/nondetermisitic-widening-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/nondetermisitic-widening-cost.ll
index 9e473b373faa8..dbd183e18b13d 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/nondetermisitic-widening-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/nondetermisitic-widening-cost.ll
@@ -15,47 +15,75 @@ define float @fun(i64 %0, float %1, ptr noalias %a, ptr noalias %b, i64 %len) #
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[TMP3:%.*]] = shl i64 [[TMP0]], 2
; CHECK-NEXT: [[TMP5:%.*]] = sub i64 [[TMP3]], 1
-; CHECK-NEXT: [[DIFF_CHECK:%.*]] = icmp ult i64 [[TMP5]], 15
+; CHECK-NEXT: [[DIFF_CHECK:%.*]] = icmp ult i64 [[TMP5]], 31
; CHECK-NEXT: br i1 [[DIFF_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr [4 x i8], ptr [[VLA]], i64 [[TMP0]]
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x float> poison, float [[TMP1]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x float> [[BROADCAST_SPLATINSERT]], <4 x float> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x float> poison, float [[TMP1]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x float> [[BROADCAST_SPLATINSERT]], <8 x float> poison, <8 x i32> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP6:%.*]] = add i64 [[INDEX]], 1
; CHECK-NEXT: [[TMP7:%.*]] = add i64 [[INDEX]], 2
; CHECK-NEXT: [[TMP8:%.*]] = add i64 [[INDEX]], 3
+; CHECK-NEXT: [[TMP12:%.*]] = add i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[INDEX]], 5
+; CHECK-NEXT: [[TMP10:%.*]] = add i64 [[INDEX]], 6
+; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[INDEX]], 7
; CHECK-NEXT: [[TMP15:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[INDEX]]
; CHECK-NEXT: [[TMP16:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP6]]
; CHECK-NEXT: [[TMP17:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP7]]
; CHECK-NEXT: [[TMP18:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP8]]
+; CHECK-NEXT: [[TMP20:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP12]]
+; CHECK-NEXT: [[TMP21:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP9]]
+; CHECK-NEXT: [[TMP22:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP10]]
+; CHECK-NEXT: [[TMP19:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP11]]
; CHECK-NEXT: [[TMP23:%.*]] = load ptr, ptr [[TMP15]], align 8
; CHECK-NEXT: [[TMP24:%.*]] = load ptr, ptr [[TMP16]], align 8
; CHECK-NEXT: [[TMP25:%.*]] = load ptr, ptr [[TMP17]], align 8
; CHECK-NEXT: [[TMP26:%.*]] = load ptr, ptr [[TMP18]], align 8
+; CHECK-NEXT: [[TMP28:%.*]] = load ptr, ptr [[TMP20]], align 8
+; CHECK-NEXT: [[TMP29:%.*]] = load ptr, ptr [[TMP21]], align 8
+; CHECK-NEXT: [[TMP30:%.*]] = load ptr, ptr [[TMP22]], align 8
+; CHECK-NEXT: [[TMP27:%.*]] = load ptr, ptr [[TMP19]], align 8
; CHECK-NEXT: [[TMP31:%.*]] = load i64, ptr [[TMP23]], align 8
; CHECK-NEXT: [[TMP32:%.*]] = load i64, ptr [[TMP24]], align 8
; CHECK-NEXT: [[TMP36:%.*]] = load i64, ptr [[TMP25]], align 8
; CHECK-NEXT: [[TMP34:%.*]] = load i64, ptr [[TMP26]], align 8
+; CHECK-NEXT: [[TMP60:%.*]] = load i64, ptr [[TMP28]], align 8
+; CHECK-NEXT: [[TMP33:%.*]] = load i64, ptr [[TMP29]], align 8
+; CHECK-NEXT: [[TMP62:%.*]] = load i64, ptr [[TMP30]], align 8
+; CHECK-NEXT: [[TMP35:%.*]] = load i64, ptr [[TMP27]], align 8
; CHECK-NEXT: [[TMP44:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP31]]
; CHECK-NEXT: [[TMP45:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP32]]
; CHECK-NEXT: [[TMP46:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP36]]
; CHECK-NEXT: [[TMP47:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP34]]
+; CHECK-NEXT: [[TMP64:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP60]]
+; CHECK-NEXT: [[TMP65:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP33]]
+; CHECK-NEXT: [[TMP66:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP62]]
+; CHECK-NEXT: [[TMP43:%.*]] = getelementptr [4 x i8], ptr [[A]], i64 [[TMP35]]
; CHECK-NEXT: [[TMP71:%.*]] = load float, ptr [[TMP44]], align 4
; CHECK-NEXT: [[TMP72:%.*]] = load float, ptr [[TMP45]], align 4
; CHECK-NEXT: [[TMP73:%.*]] = load float, ptr [[TMP46]], align 4
; CHECK-NEXT: [[TMP74:%.*]] = load float, ptr [[TMP47]], align 4
-; CHECK-NEXT: [[TMP75:%.*]] = insertelement <4 x float> poison, float [[TMP71]], i64 0
-; CHECK-NEXT: [[TMP76:%.*]] = insertelement <4 x float> [[TMP75]], float [[TMP72]], i64 1
-; CHECK-NEXT: [[TMP77:%.*]] = insertelement <4 x float> [[TMP76]], float [[TMP73]], i64 2
-; CHECK-NEXT: [[TMP78:%.*]] = insertelement <4 x float> [[TMP77]], float [[TMP74]], i64 3
+; CHECK-NEXT: [[TMP48:%.*]] = load float, ptr [[TMP64]], align 4
+; CHECK-NEXT: [[TMP49:%.*]] = load float, ptr [[TMP65]], align 4
+; CHECK-NEXT: [[TMP50:%.*]] = load float, ptr [[TMP66]], align 4
+; CHECK-NEXT: [[TMP51:%.*]] = load float, ptr [[TMP43]], align 4
+; CHECK-NEXT: [[TMP52:%.*]] = insertelement <8 x float> poison, float [[TMP71]], i64 0
+; CHECK-NEXT: [[TMP53:%.*]] = insertelement <8 x float> [[TMP52]], float [[TMP72]], i64 1
+; CHECK-NEXT: [[TMP54:%.*]] = insertelement <8 x float> [[TMP53]], float [[TMP73]], i64 2
+; CHECK-NEXT: [[TMP55:%.*]] = insertelement <8 x float> [[TMP54]], float [[TMP74]], i64 3
+; CHECK-NEXT: [[TMP56:%.*]] = insertelement <8 x float> [[TMP55]], float [[TMP48]], i64 4
+; CHECK-NEXT: [[TMP57:%.*]] = insertelement <8 x float> [[TMP56]], float [[TMP49]], i64 5
+; CHECK-NEXT: [[TMP58:%.*]] = insertelement <8 x float> [[TMP57]], float [[TMP50]], i64 6
+; CHECK-NEXT: [[TMP59:%.*]] = insertelement <8 x float> [[TMP58]], float [[TMP51]], i64 7
; CHECK-NEXT: [[TMP61:%.*]] = getelementptr [4 x i8], ptr [[VLA]], i64 [[INDEX]]
-; CHECK-NEXT: store <4 x float> [[TMP78]], ptr [[TMP61]], align 4
+; CHECK-NEXT: store <8 x float> [[TMP59]], ptr [[TMP61]], align 4
; CHECK-NEXT: [[TMP63:%.*]] = getelementptr [4 x i8], ptr [[TMP4]], i64 [[INDEX]]
-; CHECK-NEXT: store <4 x float> [[BROADCAST_SPLAT]], ptr [[TMP63]], align 4
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: store <8 x float> [[BROADCAST_SPLAT]], ptr [[TMP63]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; CHECK-NEXT: [[TMP37:%.*]] = icmp eq i64 [[INDEX_NEXT]], 128
; CHECK-NEXT: br i1 [[TMP37]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/pr131359-dead-for-splice.ll b/llvm/test/Transforms/LoopVectorize/X86/pr131359-dead-for-splice.ll
index 91958b6e74529..07b327180e33f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/pr131359-dead-for-splice.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/pr131359-dead-for-splice.ll
@@ -9,29 +9,14 @@ target triple = "x86_64"
define void @no_use() {
; CHECK-LABEL: define void @no_use() {
-; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
-; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
-; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[STEP_ADD:%.*]] = add nuw <4 x i32> [[VEC_IND]], splat (i32 4)
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 8
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[STEP_ADD]], splat (i32 4)
-; CHECK-NEXT: [[TMP0:%.*]] = icmp eq i32 [[INDEX_NEXT]], 40
-; CHECK-NEXT: br i1 [[TMP0]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: [[VECTOR_RECUR_EXTRACT:%.*]] = extractelement <4 x i32> [[STEP_ADD]], i64 3
-; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
-; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[SCALAR_PH:.*]]:
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[FOR:%.*]] = phi i32 [ [[VECTOR_RECUR_EXTRACT]], %[[SCALAR_PH]] ], [ [[E_0_I:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[E_0_I]] = phi i32 [ 40, %[[SCALAR_PH]] ], [ [[INC_I:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[FOR:%.*]] = phi i32 [ 0, %[[SCALAR_PH]] ], [ [[E_0_I:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[E_0_I]] = phi i32 [ 0, %[[SCALAR_PH]] ], [ [[INC_I:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[INC_I]] = add i32 [[E_0_I]], 1
; CHECK-NEXT: [[EXITCOND_NOT_I:%.*]] = icmp eq i32 [[E_0_I]], 43
-; CHECK-NEXT: br i1 [[EXITCOND_NOT_I]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND_NOT_I]], label %[[EXIT:.*]], label %[[LOOP]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -51,30 +36,15 @@ exit:
define void @dead_use() {
; CHECK-LABEL: define void @dead_use() {
-; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
-; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
-; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[STEP_ADD:%.*]] = add nuw <4 x i32> [[VEC_IND]], splat (i32 4)
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 8
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[STEP_ADD]], splat (i32 4)
-; CHECK-NEXT: [[TMP0:%.*]] = icmp eq i32 [[INDEX_NEXT]], 40
-; CHECK-NEXT: br i1 [[TMP0]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
-; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: [[VECTOR_RECUR_EXTRACT:%.*]] = extractelement <4 x i32> [[STEP_ADD]], i64 3
-; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
-; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[SCALAR_PH:.*]]:
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[D_0_I:%.*]] = phi i32 [ [[VECTOR_RECUR_EXTRACT]], %[[SCALAR_PH]] ], [ [[E_0_I:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[E_0_I]] = phi i32 [ 40, %[[SCALAR_PH]] ], [ [[INC_I:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[D_0_I:%.*]] = phi i32 [ 0, %[[SCALAR_PH]] ], [ [[E_0_I:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[E_0_I]] = phi i32 [ 0, %[[SCALAR_PH]] ], [ [[INC_I:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[DEAD:%.*]] = add i32 [[D_0_I]], 1
; CHECK-NEXT: [[INC_I]] = add i32 [[E_0_I]], 1
; CHECK-NEXT: [[EXITCOND_NOT_I:%.*]] = icmp eq i32 [[E_0_I]], 43
-; CHECK-NEXT: br i1 [[EXITCOND_NOT_I]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND_NOT_I]], label %[[EXIT:.*]], label %[[LOOP]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
diff --git a/llvm/test/Transforms/LoopVectorize/X86/pr47437.ll b/llvm/test/Transforms/LoopVectorize/X86/pr47437.ll
index e546743472b71..533e35d10d73a 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/pr47437.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/pr47437.ll
@@ -333,45 +333,82 @@ define void @test_muladd(ptr noalias nocapture %d1, ptr noalias nocapture readon
; AVX2-NEXT: entry:
; AVX2-NEXT: [[CMP30:%.*]] = icmp sgt i32 [[N:%.*]], 0
; AVX2-NEXT: br i1 [[CMP30]], label [[FOR_BODY_PREHEADER:%.*]], label [[FOR_END:%.*]]
-; AVX2: for.body.preheader:
+; AVX2: iter.check:
; AVX2-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[N]] to i64
-; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
+; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_PH:%.*]]
+; AVX2: vector.main.loop.iter.check:
+; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
+; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK1]], label [[VEC_EPILOG_PH:%.*]], label [[VECTOR_PH1:%.*]]
; AVX2: vector.ph:
-; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
+; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
; AVX2-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX2-NEXT: br label [[VECTOR_BODY:%.*]]
; AVX2: vector.body:
-; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH1]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP1:%.*]] = shl nuw nsw i64 [[INDEX]], 1
; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i16, ptr [[S1:%.*]], i64 [[TMP1]]
-; AVX2-NEXT: [[WIDE_VEC:%.*]] = load <16 x i16>, ptr [[TMP2]], align 2
-; AVX2-NEXT: [[STRIDED_VEC:%.*]] = shufflevector <16 x i16> [[WIDE_VEC]], <16 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
-; AVX2-NEXT: [[STRIDED_VEC1:%.*]] = shufflevector <16 x i16> [[WIDE_VEC]], <16 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
-; AVX2-NEXT: [[TMP4:%.*]] = sext <8 x i16> [[STRIDED_VEC]] to <8 x i32>
+; AVX2-NEXT: [[WIDE_VEC:%.*]] = load <32 x i16>, ptr [[TMP2]], align 2
+; AVX2-NEXT: [[STRIDED_VEC:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30>
+; AVX2-NEXT: [[STRIDED_VEC2:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <16 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
+; AVX2-NEXT: [[TMP3:%.*]] = sext <16 x i16> [[STRIDED_VEC]] to <16 x i32>
; AVX2-NEXT: [[TMP5:%.*]] = getelementptr inbounds i16, ptr [[S2:%.*]], i64 [[TMP1]]
-; AVX2-NEXT: [[WIDE_VEC2:%.*]] = load <16 x i16>, ptr [[TMP5]], align 2
-; AVX2-NEXT: [[STRIDED_VEC3:%.*]] = shufflevector <16 x i16> [[WIDE_VEC2]], <16 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
-; AVX2-NEXT: [[STRIDED_VEC4:%.*]] = shufflevector <16 x i16> [[WIDE_VEC2]], <16 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
-; AVX2-NEXT: [[TMP7:%.*]] = sext <8 x i16> [[STRIDED_VEC3]] to <8 x i32>
-; AVX2-NEXT: [[TMP8:%.*]] = mul nsw <8 x i32> [[TMP7]], [[TMP4]]
-; AVX2-NEXT: [[TMP9:%.*]] = sext <8 x i16> [[STRIDED_VEC1]] to <8 x i32>
-; AVX2-NEXT: [[TMP10:%.*]] = sext <8 x i16> [[STRIDED_VEC4]] to <8 x i32>
-; AVX2-NEXT: [[TMP11:%.*]] = mul nsw <8 x i32> [[TMP10]], [[TMP9]]
-; AVX2-NEXT: [[TMP12:%.*]] = add nsw <8 x i32> [[TMP11]], [[TMP8]]
+; AVX2-NEXT: [[WIDE_VEC3:%.*]] = load <32 x i16>, ptr [[TMP5]], align 2
+; AVX2-NEXT: [[STRIDED_VEC4:%.*]] = shufflevector <32 x i16> [[WIDE_VEC3]], <32 x i16> poison, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30>
+; AVX2-NEXT: [[STRIDED_VEC5:%.*]] = shufflevector <32 x i16> [[WIDE_VEC3]], <32 x i16> poison, <16 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
+; AVX2-NEXT: [[TMP11:%.*]] = sext <16 x i16> [[STRIDED_VEC4]] to <16 x i32>
+; AVX2-NEXT: [[TMP6:%.*]] = mul nsw <16 x i32> [[TMP11]], [[TMP3]]
+; AVX2-NEXT: [[TMP7:%.*]] = sext <16 x i16> [[STRIDED_VEC2]] to <16 x i32>
+; AVX2-NEXT: [[TMP8:%.*]] = sext <16 x i16> [[STRIDED_VEC5]] to <16 x i32>
+; AVX2-NEXT: [[TMP9:%.*]] = mul nsw <16 x i32> [[TMP8]], [[TMP7]]
+; AVX2-NEXT: [[TMP10:%.*]] = add nsw <16 x i32> [[TMP9]], [[TMP6]]
; AVX2-NEXT: [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[D1:%.*]], i64 [[INDEX]]
-; AVX2-NEXT: store <8 x i32> [[TMP12]], ptr [[TMP13]], align 4
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; AVX2-NEXT: store <16 x i32> [[TMP10]], ptr [[TMP13]], align 4
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX2-NEXT: [[TMP15:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX2-NEXT: br i1 [[TMP15]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; AVX2: middle.block:
; AVX2-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
-; AVX2-NEXT: br i1 [[CMP_N]], label [[FOR_END_LOOPEXIT:%.*]], label [[SCALAR_PH]]
-; AVX2: scalar.ph:
-; AVX2-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[MIDDLE_BLOCK]] ], [ 0, [[FOR_BODY_PREHEADER]] ]
+; AVX2-NEXT: br i1 [[CMP_N]], label [[FOR_END_LOOPEXIT:%.*]], label [[VEC_EPILOG_ITER_CHECK:%.*]]
+; AVX2: vec.epilog.iter.check:
+; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
+; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label [[SCALAR_PH]], label [[VEC_EPILOG_PH]], !prof [[PROF3:![0-9]+]]
+; AVX2: vec.epilog.ph:
+; AVX2-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[VECTOR_PH]] ]
+; AVX2-NEXT: [[TMP26:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
+; AVX2-NEXT: [[N_VEC6:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[TMP26]]
; AVX2-NEXT: br label [[FOR_BODY:%.*]]
+; AVX2: vec.epilog.vector.body:
+; AVX2-NEXT: [[INDEX7:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], [[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT14:%.*]], [[FOR_BODY]] ]
+; AVX2-NEXT: [[TMP14:%.*]] = shl nuw nsw i64 [[INDEX7]], 1
+; AVX2-NEXT: [[TMP27:%.*]] = getelementptr inbounds i16, ptr [[S1]], i64 [[TMP14]]
+; AVX2-NEXT: [[WIDE_VEC8:%.*]] = load <8 x i16>, ptr [[TMP27]], align 2
+; AVX2-NEXT: [[STRIDED_VEC9:%.*]] = shufflevector <8 x i16> [[WIDE_VEC8]], <8 x i16> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; AVX2-NEXT: [[STRIDED_VEC10:%.*]] = shufflevector <8 x i16> [[WIDE_VEC8]], <8 x i16> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; AVX2-NEXT: [[TMP28:%.*]] = sext <4 x i16> [[STRIDED_VEC9]] to <4 x i32>
+; AVX2-NEXT: [[TMP29:%.*]] = getelementptr inbounds i16, ptr [[S2]], i64 [[TMP14]]
+; AVX2-NEXT: [[WIDE_VEC11:%.*]] = load <8 x i16>, ptr [[TMP29]], align 2
+; AVX2-NEXT: [[STRIDED_VEC12:%.*]] = shufflevector <8 x i16> [[WIDE_VEC11]], <8 x i16> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; AVX2-NEXT: [[STRIDED_VEC13:%.*]] = shufflevector <8 x i16> [[WIDE_VEC11]], <8 x i16> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; AVX2-NEXT: [[TMP30:%.*]] = sext <4 x i16> [[STRIDED_VEC12]] to <4 x i32>
+; AVX2-NEXT: [[TMP31:%.*]] = mul nsw <4 x i32> [[TMP30]], [[TMP28]]
+; AVX2-NEXT: [[TMP32:%.*]] = sext <4 x i16> [[STRIDED_VEC10]] to <4 x i32>
+; AVX2-NEXT: [[TMP33:%.*]] = sext <4 x i16> [[STRIDED_VEC13]] to <4 x i32>
+; AVX2-NEXT: [[TMP22:%.*]] = mul nsw <4 x i32> [[TMP33]], [[TMP32]]
+; AVX2-NEXT: [[TMP23:%.*]] = add nsw <4 x i32> [[TMP22]], [[TMP31]]
+; AVX2-NEXT: [[TMP24:%.*]] = getelementptr inbounds i32, ptr [[D1]], i64 [[INDEX7]]
+; AVX2-NEXT: store <4 x i32> [[TMP23]], ptr [[TMP24]], align 4
+; AVX2-NEXT: [[INDEX_NEXT14]] = add nuw i64 [[INDEX7]], 4
+; AVX2-NEXT: [[TMP25:%.*]] = icmp eq i64 [[INDEX_NEXT14]], [[N_VEC6]]
+; AVX2-NEXT: br i1 [[TMP25]], label [[VEC_EPILOG_MIDDLE_BLOCK:%.*]], label [[FOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; AVX2: vec.epilog.middle.block:
+; AVX2-NEXT: [[CMP_N15:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC6]]
+; AVX2-NEXT: br i1 [[CMP_N15]], label [[FOR_END_LOOPEXIT]], label [[SCALAR_PH]]
+; AVX2: vec.epilog.scalar.ph:
+; AVX2-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC6]], [[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[FOR_BODY_PREHEADER]] ]
+; AVX2-NEXT: br label [[FOR_BODY1:%.*]]
; AVX2: for.body:
-; AVX2-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], [[SCALAR_PH]] ], [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY]] ]
+; AVX2-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], [[SCALAR_PH]] ], [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY1]] ]
; AVX2-NEXT: [[TMP16:%.*]] = shl nuw nsw i64 [[INDVARS_IV]], 1
; AVX2-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds i16, ptr [[S1]], i64 [[TMP16]]
; AVX2-NEXT: [[TMP17:%.*]] = load i16, ptr [[ARRAYIDX]], align 2
@@ -393,7 +430,7 @@ define void @test_muladd(ptr noalias nocapture %d1, ptr noalias nocapture readon
; AVX2-NEXT: store i32 [[ADD18]], ptr [[ARRAYIDX20]], align 4
; AVX2-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1
; AVX2-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], [[WIDE_TRIP_COUNT]]
-; AVX2-NEXT: br i1 [[EXITCOND_NOT]], label [[FOR_END_LOOPEXIT]], label [[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; AVX2-NEXT: br i1 [[EXITCOND_NOT]], label [[FOR_END_LOOPEXIT]], label [[FOR_BODY1]], !llvm.loop [[LOOP5:![0-9]+]]
; AVX2: for.end.loopexit:
; AVX2-NEXT: br label [[FOR_END]]
; AVX2: for.end:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/reduction-crash.ll b/llvm/test/Transforms/LoopVectorize/X86/reduction-crash.ll
index 65759d545e4af..6c61d2fb9b949 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/reduction-crash.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/reduction-crash.ll
@@ -12,7 +12,7 @@ define void @pr15344(ptr noalias %ar, ptr noalias %ar2, i32 %exit.limit, i1 %con
; CHECK: [[PH]]:
; CHECK-NEXT: br i1 [[COND]], label %[[LOOP_PREHEADER:.*]], label %[[EXIT:.*]]
; CHECK: [[LOOP_PREHEADER]]:
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[EXIT_LIMIT]], 10
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[EXIT_LIMIT]], 12
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[TMP0:%.*]] = shl i32 [[EXIT_LIMIT]], 2
@@ -24,25 +24,25 @@ define void @pr15344(ptr noalias %ar, ptr noalias %ar2, i32 %exit.limit, i1 %con
; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
; CHECK-NEXT: br i1 [[FOUND_CONFLICT]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i32 [[EXIT_LIMIT]], 3
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i32 [[EXIT_LIMIT]], 7
; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[EXIT_LIMIT]], [[N_MOD_VF]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <2 x double> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP2:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI2:%.*]] = phi <2 x double> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP3:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP2]] = fadd fast <2 x double> [[VEC_PHI]], splat (double 1.000000e+00)
-; CHECK-NEXT: [[TMP3]] = fadd fast <2 x double> [[VEC_PHI2]], splat (double 1.000000e+00)
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x double> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP3:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI2:%.*]] = phi <4 x double> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP3]] = fadd fast <4 x double> [[VEC_PHI]], splat (double 1.000000e+00)
+; CHECK-NEXT: [[TMP5]] = fadd fast <4 x double> [[VEC_PHI2]], splat (double 1.000000e+00)
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds float, ptr [[AR2]], i32 [[INDEX]]
-; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds float, ptr [[TMP4]], i32 2
-; CHECK-NEXT: store <2 x float> splat (float 2.000000e+00), ptr [[TMP4]], align 4, !alias.scope [[META0:![0-9]+]], !noalias [[META3:![0-9]+]]
-; CHECK-NEXT: store <2 x float> splat (float 2.000000e+00), ptr [[TMP6]], align 4, !alias.scope [[META0]], !noalias [[META3]]
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds float, ptr [[TMP4]], i32 4
+; CHECK-NEXT: store <4 x float> splat (float 2.000000e+00), ptr [[TMP4]], align 4, !alias.scope [[META0:![0-9]+]], !noalias [[META3:![0-9]+]]
+; CHECK-NEXT: store <4 x float> splat (float 2.000000e+00), ptr [[TMP6]], align 4, !alias.scope [[META0]], !noalias [[META3]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 8
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: [[BIN_RDX:%.*]] = fadd fast <2 x double> [[TMP3]], [[TMP2]]
-; CHECK-NEXT: [[TMP8:%.*]] = call fast double @llvm.vector.reduce.fadd.v2f64(double 0.000000e+00, <2 x double> [[BIN_RDX]])
+; CHECK-NEXT: [[BIN_RDX:%.*]] = fadd fast <4 x double> [[TMP5]], [[TMP3]]
+; CHECK-NEXT: [[TMP8:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[BIN_RDX]])
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i32 [[EXIT_LIMIT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT_LOOPEXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/replicating-load-store-costs.ll b/llvm/test/Transforms/LoopVectorize/X86/replicating-load-store-costs.ll
index 35c58e0880402..8976d605a19e0 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/replicating-load-store-costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/replicating-load-store-costs.ll
@@ -165,23 +165,17 @@ define void @test_store_initially_interleave(i32 %n, ptr noalias %src) #0 {
; I32-SAME: i32 [[N:%.*]], ptr noalias [[SRC:%.*]]) #[[ATTR0:[0-9]+]] {
; I32-NEXT: [[ITER_CHECK:.*:]]
; I32-NEXT: [[TMP0:%.*]] = add i32 [[N]], 1
-; I32-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ule i32 [[TMP0]], 4
-; I32-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; I32: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; I32-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ule i32 [[TMP0]], 16
+; I32-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ule i32 [[TMP0]], 8
; I32-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; I32: [[VECTOR_PH]]:
-; I32-NEXT: [[N_MOD_VF:%.*]] = and i32 [[TMP0]], 15
+; I32-NEXT: [[N_MOD_VF:%.*]] = and i32 [[TMP0]], 7
; I32-NEXT: [[TMP1:%.*]] = icmp eq i32 [[N_MOD_VF]], 0
-; I32-NEXT: [[TMP2:%.*]] = select i1 [[TMP1]], i32 16, i32 [[N_MOD_VF]]
+; I32-NEXT: [[TMP2:%.*]] = select i1 [[TMP1]], i32 8, i32 [[N_MOD_VF]]
; I32-NEXT: [[N_VEC:%.*]] = sub i32 [[TMP0]], [[TMP2]]
; I32-NEXT: br label %[[VECTOR_BODY:.*]]
; I32: [[VECTOR_BODY]]:
; I32-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; I32-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; I32-NEXT: [[STEP_ADD:%.*]] = add nuw <4 x i32> [[VEC_IND]], splat (i32 4)
-; I32-NEXT: [[STEP_ADD_2:%.*]] = add nuw <4 x i32> [[STEP_ADD]], splat (i32 4)
-; I32-NEXT: [[STEP_ADD_3:%.*]] = add nuw <4 x i32> [[STEP_ADD_2]], splat (i32 4)
+; I32-NEXT: [[VEC_IND:%.*]] = phi <8 x i32> [ <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
; I32-NEXT: [[TMP3:%.*]] = add i32 [[INDEX]], 1
; I32-NEXT: [[TMP4:%.*]] = add i32 [[INDEX]], 2
; I32-NEXT: [[TMP5:%.*]] = add i32 [[INDEX]], 3
@@ -189,131 +183,46 @@ define void @test_store_initially_interleave(i32 %n, ptr noalias %src) #0 {
; I32-NEXT: [[TMP7:%.*]] = add i32 [[INDEX]], 5
; I32-NEXT: [[TMP8:%.*]] = add i32 [[INDEX]], 6
; I32-NEXT: [[TMP9:%.*]] = add i32 [[INDEX]], 7
-; I32-NEXT: [[TMP10:%.*]] = add i32 [[INDEX]], 8
-; I32-NEXT: [[TMP11:%.*]] = add i32 [[INDEX]], 9
-; I32-NEXT: [[TMP12:%.*]] = add i32 [[INDEX]], 10
-; I32-NEXT: [[TMP13:%.*]] = add i32 [[INDEX]], 11
-; I32-NEXT: [[TMP14:%.*]] = add i32 [[INDEX]], 12
-; I32-NEXT: [[TMP15:%.*]] = add i32 [[INDEX]], 13
-; I32-NEXT: [[TMP16:%.*]] = add i32 [[INDEX]], 14
-; I32-NEXT: [[TMP17:%.*]] = add i32 [[INDEX]], 15
-; I32-NEXT: [[TMP18:%.*]] = uitofp <4 x i32> [[VEC_IND]] to <4 x double>
-; I32-NEXT: [[TMP23:%.*]] = uitofp <4 x i32> [[STEP_ADD]] to <4 x double>
-; I32-NEXT: [[TMP28:%.*]] = uitofp <4 x i32> [[STEP_ADD_2]] to <4 x double>
-; I32-NEXT: [[TMP33:%.*]] = uitofp <4 x i32> [[STEP_ADD_3]] to <4 x double>
-; I32-NEXT: [[TMP53:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[INDEX]]
-; I32-NEXT: [[TMP54:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP3]]
-; I32-NEXT: [[TMP55:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP4]]
-; I32-NEXT: [[TMP56:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP5]]
-; I32-NEXT: [[TMP57:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP6]]
-; I32-NEXT: [[TMP58:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP7]]
-; I32-NEXT: [[TMP59:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP8]]
-; I32-NEXT: [[TMP60:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP9]]
-; I32-NEXT: [[TMP61:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP10]]
-; I32-NEXT: [[TMP62:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP11]]
-; I32-NEXT: [[TMP63:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP12]]
-; I32-NEXT: [[TMP64:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP13]]
-; I32-NEXT: [[TMP65:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP14]]
-; I32-NEXT: [[TMP66:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP15]]
-; I32-NEXT: [[TMP67:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP16]]
-; I32-NEXT: [[TMP68:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP17]]
-; I32-NEXT: [[TMP38:%.*]] = load ptr, ptr [[TMP53]], align 4
-; I32-NEXT: [[TMP39:%.*]] = load ptr, ptr [[TMP54]], align 4
-; I32-NEXT: [[TMP40:%.*]] = load ptr, ptr [[TMP55]], align 4
-; I32-NEXT: [[TMP41:%.*]] = load ptr, ptr [[TMP56]], align 4
-; I32-NEXT: [[TMP42:%.*]] = load ptr, ptr [[TMP57]], align 4
-; I32-NEXT: [[TMP43:%.*]] = load ptr, ptr [[TMP58]], align 4
-; I32-NEXT: [[TMP44:%.*]] = load ptr, ptr [[TMP59]], align 4
-; I32-NEXT: [[TMP45:%.*]] = load ptr, ptr [[TMP60]], align 4
-; I32-NEXT: [[TMP46:%.*]] = load ptr, ptr [[TMP61]], align 4
-; I32-NEXT: [[TMP47:%.*]] = load ptr, ptr [[TMP62]], align 4
-; I32-NEXT: [[TMP48:%.*]] = load ptr, ptr [[TMP63]], align 4
-; I32-NEXT: [[TMP49:%.*]] = load ptr, ptr [[TMP64]], align 4
+; I32-NEXT: [[TMP11:%.*]] = uitofp <8 x i32> [[VEC_IND]] to <8 x double>
+; I32-NEXT: [[TMP65:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[INDEX]]
+; I32-NEXT: [[TMP66:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP3]]
+; I32-NEXT: [[TMP67:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP4]]
+; I32-NEXT: [[TMP68:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP5]]
+; I32-NEXT: [[TMP84:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP6]]
+; I32-NEXT: [[TMP85:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP7]]
+; I32-NEXT: [[TMP86:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP8]]
+; I32-NEXT: [[TMP87:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP9]]
; I32-NEXT: [[TMP50:%.*]] = load ptr, ptr [[TMP65]], align 4
; I32-NEXT: [[TMP51:%.*]] = load ptr, ptr [[TMP66]], align 4
; I32-NEXT: [[TMP52:%.*]] = load ptr, ptr [[TMP67]], align 4
; I32-NEXT: [[TMP69:%.*]] = load ptr, ptr [[TMP68]], align 4
-; I32-NEXT: [[TMP19:%.*]] = extractelement <4 x double> [[TMP18]], i64 0
-; I32-NEXT: store double [[TMP19]], ptr [[TMP38]], align 4
-; I32-NEXT: [[TMP20:%.*]] = extractelement <4 x double> [[TMP18]], i64 1
-; I32-NEXT: store double [[TMP20]], ptr [[TMP39]], align 4
-; I32-NEXT: [[TMP21:%.*]] = extractelement <4 x double> [[TMP18]], i64 2
-; I32-NEXT: store double [[TMP21]], ptr [[TMP40]], align 4
-; I32-NEXT: [[TMP22:%.*]] = extractelement <4 x double> [[TMP18]], i64 3
-; I32-NEXT: store double [[TMP22]], ptr [[TMP41]], align 4
-; I32-NEXT: [[TMP24:%.*]] = extractelement <4 x double> [[TMP23]], i64 0
-; I32-NEXT: store double [[TMP24]], ptr [[TMP42]], align 4
-; I32-NEXT: [[TMP25:%.*]] = extractelement <4 x double> [[TMP23]], i64 1
-; I32-NEXT: store double [[TMP25]], ptr [[TMP43]], align 4
-; I32-NEXT: [[TMP26:%.*]] = extractelement <4 x double> [[TMP23]], i64 2
-; I32-NEXT: store double [[TMP26]], ptr [[TMP44]], align 4
-; I32-NEXT: [[TMP27:%.*]] = extractelement <4 x double> [[TMP23]], i64 3
-; I32-NEXT: store double [[TMP27]], ptr [[TMP45]], align 4
-; I32-NEXT: [[TMP29:%.*]] = extractelement <4 x double> [[TMP28]], i64 0
-; I32-NEXT: store double [[TMP29]], ptr [[TMP46]], align 4
-; I32-NEXT: [[TMP30:%.*]] = extractelement <4 x double> [[TMP28]], i64 1
-; I32-NEXT: store double [[TMP30]], ptr [[TMP47]], align 4
-; I32-NEXT: [[TMP31:%.*]] = extractelement <4 x double> [[TMP28]], i64 2
-; I32-NEXT: store double [[TMP31]], ptr [[TMP48]], align 4
-; I32-NEXT: [[TMP32:%.*]] = extractelement <4 x double> [[TMP28]], i64 3
-; I32-NEXT: store double [[TMP32]], ptr [[TMP49]], align 4
-; I32-NEXT: [[TMP34:%.*]] = extractelement <4 x double> [[TMP33]], i64 0
-; I32-NEXT: store double [[TMP34]], ptr [[TMP50]], align 4
-; I32-NEXT: [[TMP35:%.*]] = extractelement <4 x double> [[TMP33]], i64 1
-; I32-NEXT: store double [[TMP35]], ptr [[TMP51]], align 4
-; I32-NEXT: [[TMP36:%.*]] = extractelement <4 x double> [[TMP33]], i64 2
-; I32-NEXT: store double [[TMP36]], ptr [[TMP52]], align 4
-; I32-NEXT: [[TMP37:%.*]] = extractelement <4 x double> [[TMP33]], i64 3
-; I32-NEXT: store double [[TMP37]], ptr [[TMP69]], align 4
-; I32-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 16
-; I32-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[STEP_ADD_3]], splat (i32 4)
-; I32-NEXT: [[TMP70:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
-; I32-NEXT: br i1 [[TMP70]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; I32: [[MIDDLE_BLOCK]]:
-; I32-NEXT: br label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; I32: [[VEC_EPILOG_ITER_CHECK]]:
-; I32-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ule i32 [[TMP2]], 4
-; I32-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3:![0-9]+]]
-; I32: [[VEC_EPILOG_PH]]:
-; I32-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; I32-NEXT: [[N_MOD_VF2:%.*]] = and i32 [[TMP0]], 3
-; I32-NEXT: [[TMP71:%.*]] = icmp eq i32 [[N_MOD_VF2]], 0
-; I32-NEXT: [[TMP72:%.*]] = select i1 [[TMP71]], i32 4, i32 [[N_MOD_VF2]]
-; I32-NEXT: [[N_VEC3:%.*]] = sub i32 [[TMP0]], [[TMP72]]
-; I32-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[VEC_EPILOG_RESUME_VAL]], i64 0
-; I32-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
-; I32-NEXT: [[INDUCTION:%.*]] = add <4 x i32> [[BROADCAST_SPLAT]], <i32 0, i32 1, i32 2, i32 3>
-; I32-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; I32: [[VEC_EPILOG_VECTOR_BODY]]:
-; I32-NEXT: [[INDEX4:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; I32-NEXT: [[VEC_IND5:%.*]] = phi <4 x i32> [ [[INDUCTION]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT7:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; I32-NEXT: [[TMP73:%.*]] = add i32 [[INDEX4]], 1
-; I32-NEXT: [[TMP74:%.*]] = add i32 [[INDEX4]], 2
-; I32-NEXT: [[TMP75:%.*]] = add i32 [[INDEX4]], 3
-; I32-NEXT: [[TMP76:%.*]] = uitofp <4 x i32> [[VEC_IND5]] to <4 x double>
-; I32-NEXT: [[TMP84:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[INDEX4]]
-; I32-NEXT: [[TMP85:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP73]]
-; I32-NEXT: [[TMP86:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP74]]
-; I32-NEXT: [[TMP87:%.*]] = getelementptr nusw { ptr, ptr, ptr }, ptr null, i32 [[TMP75]]
; I32-NEXT: [[TMP81:%.*]] = load ptr, ptr [[TMP84]], align 4
; I32-NEXT: [[TMP82:%.*]] = load ptr, ptr [[TMP85]], align 4
; I32-NEXT: [[TMP83:%.*]] = load ptr, ptr [[TMP86]], align 4
; I32-NEXT: [[TMP88:%.*]] = load ptr, ptr [[TMP87]], align 4
-; I32-NEXT: [[TMP77:%.*]] = extractelement <4 x double> [[TMP76]], i64 0
+; I32-NEXT: [[TMP28:%.*]] = extractelement <8 x double> [[TMP11]], i64 0
+; I32-NEXT: store double [[TMP28]], ptr [[TMP50]], align 4
+; I32-NEXT: [[TMP29:%.*]] = extractelement <8 x double> [[TMP11]], i64 1
+; I32-NEXT: store double [[TMP29]], ptr [[TMP51]], align 4
+; I32-NEXT: [[TMP30:%.*]] = extractelement <8 x double> [[TMP11]], i64 2
+; I32-NEXT: store double [[TMP30]], ptr [[TMP52]], align 4
+; I32-NEXT: [[TMP31:%.*]] = extractelement <8 x double> [[TMP11]], i64 3
+; I32-NEXT: store double [[TMP31]], ptr [[TMP69]], align 4
+; I32-NEXT: [[TMP77:%.*]] = extractelement <8 x double> [[TMP11]], i64 4
; I32-NEXT: store double [[TMP77]], ptr [[TMP81]], align 4
-; I32-NEXT: [[TMP78:%.*]] = extractelement <4 x double> [[TMP76]], i64 1
+; I32-NEXT: [[TMP78:%.*]] = extractelement <8 x double> [[TMP11]], i64 5
; I32-NEXT: store double [[TMP78]], ptr [[TMP82]], align 4
-; I32-NEXT: [[TMP79:%.*]] = extractelement <4 x double> [[TMP76]], i64 2
+; I32-NEXT: [[TMP79:%.*]] = extractelement <8 x double> [[TMP11]], i64 6
; I32-NEXT: store double [[TMP79]], ptr [[TMP83]], align 4
-; I32-NEXT: [[TMP80:%.*]] = extractelement <4 x double> [[TMP76]], i64 3
+; I32-NEXT: [[TMP80:%.*]] = extractelement <8 x double> [[TMP11]], i64 7
; I32-NEXT: store double [[TMP80]], ptr [[TMP88]], align 4
-; I32-NEXT: [[INDEX_NEXT6]] = add nuw i32 [[INDEX4]], 4
-; I32-NEXT: [[VEC_IND_NEXT7]] = add <4 x i32> [[VEC_IND5]], splat (i32 4)
-; I32-NEXT: [[TMP89:%.*]] = icmp eq i32 [[INDEX_NEXT6]], [[N_VEC3]]
-; I32-NEXT: br i1 [[TMP89]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; I32-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 8
+; I32-NEXT: [[VEC_IND_NEXT]] = add <8 x i32> [[VEC_IND]], splat (i32 8)
+; I32-NEXT: [[TMP36:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; I32-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; I32: [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; I32-NEXT: br label %[[VEC_EPILOG_SCALAR_PH]]
-; I32: [[VEC_EPILOG_SCALAR_PH]]:
+; I32-NEXT: br label %[[VEC_EPILOG_PH]]
+; I32: [[VEC_EPILOG_PH]]:
;
entry:
br label %loop
@@ -419,7 +328,7 @@ define void @test_store_loaded_value(ptr noalias %src, ptr noalias %dst, i32 %n)
; I32-NEXT: store double [[TMP10]], ptr [[TMP18]], align 8
; I32-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; I32-NEXT: [[TMP19:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; I32-NEXT: br i1 [[TMP19]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; I32-NEXT: br i1 [[TMP19]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
; I32: [[MIDDLE_BLOCK]]:
; I32-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N_EXT]], [[N_VEC]]
; I32-NEXT: br i1 [[CMP_N]], [[EXIT_LOOPEXIT:label %.*]], label %[[SCALAR_PH]]
@@ -817,7 +726,7 @@ define void @loaded_address_used_by_load_through_blend(i64 %start, ptr noalias %
; I32-NEXT: store float [[TMP82]], ptr [[TMP90]], align 4
; I32-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; I32-NEXT: [[TMP91:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; I32-NEXT: br i1 [[TMP91]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
+; I32-NEXT: br i1 [[TMP91]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
; I32: [[MIDDLE_BLOCK]]:
; I32-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP1]], [[N_VEC]]
; I32-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
@@ -868,54 +777,102 @@ define void @address_use_in_different_block(ptr noalias %dst, ptr %src.0, ptr %s
; I64-NEXT: [[TMP0:%.*]] = add i64 [[INDEX]], 1
; I64-NEXT: [[TMP1:%.*]] = add i64 [[INDEX]], 2
; I64-NEXT: [[TMP2:%.*]] = add i64 [[INDEX]], 3
+; I64-NEXT: [[TMP3:%.*]] = add i64 [[INDEX]], 4
+; I64-NEXT: [[TMP4:%.*]] = add i64 [[INDEX]], 5
+; I64-NEXT: [[TMP5:%.*]] = add i64 [[INDEX]], 6
+; I64-NEXT: [[TMP6:%.*]] = add i64 [[INDEX]], 7
; I64-NEXT: [[TMP11:%.*]] = mul i64 [[INDEX]], [[OFFSET]]
; I64-NEXT: [[TMP12:%.*]] = mul i64 [[TMP0]], [[OFFSET]]
; I64-NEXT: [[TMP13:%.*]] = mul i64 [[TMP1]], [[OFFSET]]
; I64-NEXT: [[TMP14:%.*]] = mul i64 [[TMP2]], [[OFFSET]]
+; I64-NEXT: [[TMP15:%.*]] = mul i64 [[TMP3]], [[OFFSET]]
+; I64-NEXT: [[TMP16:%.*]] = mul i64 [[TMP4]], [[OFFSET]]
+; I64-NEXT: [[TMP17:%.*]] = mul i64 [[TMP5]], [[OFFSET]]
+; I64-NEXT: [[TMP18:%.*]] = mul i64 [[TMP6]], [[OFFSET]]
; I64-NEXT: [[TMP19:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP11]]
; I64-NEXT: [[TMP20:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP12]]
; I64-NEXT: [[TMP21:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP13]]
; I64-NEXT: [[TMP22:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP14]]
+; I64-NEXT: [[TMP23:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP15]]
+; I64-NEXT: [[TMP24:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP16]]
+; I64-NEXT: [[TMP25:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP17]]
+; I64-NEXT: [[TMP26:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP18]]
; I64-NEXT: [[TMP27:%.*]] = load i32, ptr [[TMP19]], align 4
; I64-NEXT: [[TMP28:%.*]] = load i32, ptr [[TMP20]], align 4
; I64-NEXT: [[TMP29:%.*]] = load i32, ptr [[TMP21]], align 4
; I64-NEXT: [[TMP30:%.*]] = load i32, ptr [[TMP22]], align 4
+; I64-NEXT: [[TMP31:%.*]] = load i32, ptr [[TMP23]], align 4
+; I64-NEXT: [[TMP32:%.*]] = load i32, ptr [[TMP24]], align 4
+; I64-NEXT: [[TMP33:%.*]] = load i32, ptr [[TMP25]], align 4
+; I64-NEXT: [[TMP34:%.*]] = load i32, ptr [[TMP26]], align 4
; I64-NEXT: [[TMP35:%.*]] = sext i32 [[TMP27]] to i64
; I64-NEXT: [[TMP36:%.*]] = sext i32 [[TMP28]] to i64
; I64-NEXT: [[TMP37:%.*]] = sext i32 [[TMP29]] to i64
; I64-NEXT: [[TMP38:%.*]] = sext i32 [[TMP30]] to i64
+; I64-NEXT: [[TMP39:%.*]] = sext i32 [[TMP31]] to i64
+; I64-NEXT: [[TMP40:%.*]] = sext i32 [[TMP32]] to i64
+; I64-NEXT: [[TMP41:%.*]] = sext i32 [[TMP33]] to i64
+; I64-NEXT: [[TMP42:%.*]] = sext i32 [[TMP34]] to i64
; I64-NEXT: [[TMP43:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP35]]
; I64-NEXT: [[TMP44:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP36]]
; I64-NEXT: [[TMP45:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP37]]
; I64-NEXT: [[TMP46:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP38]]
+; I64-NEXT: [[TMP47:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP39]]
+; I64-NEXT: [[TMP48:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP40]]
+; I64-NEXT: [[TMP49:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP41]]
+; I64-NEXT: [[TMP50:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP42]]
; I64-NEXT: [[TMP51:%.*]] = getelementptr i8, ptr [[TMP43]], i64 -8
; I64-NEXT: [[TMP52:%.*]] = getelementptr i8, ptr [[TMP44]], i64 -8
; I64-NEXT: [[TMP53:%.*]] = getelementptr i8, ptr [[TMP45]], i64 -8
; I64-NEXT: [[TMP54:%.*]] = getelementptr i8, ptr [[TMP46]], i64 -8
+; I64-NEXT: [[TMP55:%.*]] = getelementptr i8, ptr [[TMP47]], i64 -8
+; I64-NEXT: [[TMP56:%.*]] = getelementptr i8, ptr [[TMP48]], i64 -8
+; I64-NEXT: [[TMP57:%.*]] = getelementptr i8, ptr [[TMP49]], i64 -8
+; I64-NEXT: [[TMP58:%.*]] = getelementptr i8, ptr [[TMP50]], i64 -8
; I64-NEXT: [[TMP63:%.*]] = load double, ptr [[TMP51]], align 8
; I64-NEXT: [[TMP64:%.*]] = load double, ptr [[TMP52]], align 8
; I64-NEXT: [[TMP67:%.*]] = load double, ptr [[TMP53]], align 8
; I64-NEXT: [[TMP68:%.*]] = load double, ptr [[TMP54]], align 8
-; I64-NEXT: [[TMP31:%.*]] = insertelement <4 x double> poison, double [[TMP63]], i64 0
-; I64-NEXT: [[TMP32:%.*]] = insertelement <4 x double> [[TMP31]], double [[TMP64]], i64 1
-; I64-NEXT: [[TMP33:%.*]] = insertelement <4 x double> [[TMP32]], double [[TMP67]], i64 2
-; I64-NEXT: [[TMP34:%.*]] = insertelement <4 x double> [[TMP33]], double [[TMP68]], i64 3
-; I64-NEXT: [[TMP39:%.*]] = fsub <4 x double> zeroinitializer, [[TMP34]]
+; I64-NEXT: [[TMP59:%.*]] = load double, ptr [[TMP55]], align 8
+; I64-NEXT: [[TMP60:%.*]] = load double, ptr [[TMP56]], align 8
+; I64-NEXT: [[TMP61:%.*]] = load double, ptr [[TMP57]], align 8
+; I64-NEXT: [[TMP62:%.*]] = load double, ptr [[TMP58]], align 8
+; I64-NEXT: [[TMP72:%.*]] = insertelement <8 x double> poison, double [[TMP63]], i64 0
+; I64-NEXT: [[TMP73:%.*]] = insertelement <8 x double> [[TMP72]], double [[TMP64]], i64 1
+; I64-NEXT: [[TMP65:%.*]] = insertelement <8 x double> [[TMP73]], double [[TMP67]], i64 2
+; I64-NEXT: [[TMP66:%.*]] = insertelement <8 x double> [[TMP65]], double [[TMP68]], i64 3
+; I64-NEXT: [[TMP74:%.*]] = insertelement <8 x double> [[TMP66]], double [[TMP59]], i64 4
+; I64-NEXT: [[TMP75:%.*]] = insertelement <8 x double> [[TMP74]], double [[TMP60]], i64 5
+; I64-NEXT: [[TMP69:%.*]] = insertelement <8 x double> [[TMP75]], double [[TMP61]], i64 6
+; I64-NEXT: [[TMP70:%.*]] = insertelement <8 x double> [[TMP69]], double [[TMP62]], i64 7
+; I64-NEXT: [[TMP71:%.*]] = fsub <8 x double> zeroinitializer, [[TMP70]]
; I64-NEXT: [[TMP87:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP11]]
; I64-NEXT: [[TMP88:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP12]]
; I64-NEXT: [[TMP89:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP13]]
; I64-NEXT: [[TMP90:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP14]]
-; I64-NEXT: [[TMP78:%.*]] = extractelement <4 x double> [[TMP39]], i64 0
+; I64-NEXT: [[TMP76:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP15]]
+; I64-NEXT: [[TMP77:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP16]]
+; I64-NEXT: [[TMP80:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP17]]
+; I64-NEXT: [[TMP83:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP18]]
+; I64-NEXT: [[TMP78:%.*]] = extractelement <8 x double> [[TMP71]], i64 0
; I64-NEXT: store double [[TMP78]], ptr [[TMP87]], align 8
-; I64-NEXT: [[TMP79:%.*]] = extractelement <4 x double> [[TMP39]], i64 1
+; I64-NEXT: [[TMP79:%.*]] = extractelement <8 x double> [[TMP71]], i64 1
; I64-NEXT: store double [[TMP79]], ptr [[TMP88]], align 8
-; I64-NEXT: [[TMP81:%.*]] = extractelement <4 x double> [[TMP39]], i64 2
+; I64-NEXT: [[TMP81:%.*]] = extractelement <8 x double> [[TMP71]], i64 2
; I64-NEXT: store double [[TMP81]], ptr [[TMP89]], align 8
-; I64-NEXT: [[TMP82:%.*]] = extractelement <4 x double> [[TMP39]], i64 3
+; I64-NEXT: [[TMP82:%.*]] = extractelement <8 x double> [[TMP71]], i64 3
; I64-NEXT: store double [[TMP82]], ptr [[TMP90]], align 8
-; I64-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
-; I64-NEXT: [[TMP47:%.*]] = icmp eq i64 [[INDEX_NEXT]], 100
-; I64-NEXT: br i1 [[TMP47]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; I64-NEXT: [[TMP84:%.*]] = extractelement <8 x double> [[TMP71]], i64 4
+; I64-NEXT: store double [[TMP84]], ptr [[TMP76]], align 8
+; I64-NEXT: [[TMP85:%.*]] = extractelement <8 x double> [[TMP71]], i64 5
+; I64-NEXT: store double [[TMP85]], ptr [[TMP77]], align 8
+; I64-NEXT: [[TMP86:%.*]] = extractelement <8 x double> [[TMP71]], i64 6
+; I64-NEXT: store double [[TMP86]], ptr [[TMP80]], align 8
+; I64-NEXT: [[TMP91:%.*]] = extractelement <8 x double> [[TMP71]], i64 7
+; I64-NEXT: store double [[TMP91]], ptr [[TMP83]], align 8
+; I64-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; I64-NEXT: [[TMP92:%.*]] = icmp eq i64 [[INDEX_NEXT]], 96
+; I64-NEXT: br i1 [[TMP92]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
; I64: [[MIDDLE_BLOCK]]:
; I64-NEXT: br label %[[SCALAR_PH:.*]]
; I64: [[SCALAR_PH]]:
@@ -933,54 +890,102 @@ define void @address_use_in_different_block(ptr noalias %dst, ptr %src.0, ptr %s
; I32-NEXT: [[TMP0:%.*]] = add i64 [[INDEX]], 1
; I32-NEXT: [[TMP1:%.*]] = add i64 [[INDEX]], 2
; I32-NEXT: [[TMP2:%.*]] = add i64 [[INDEX]], 3
+; I32-NEXT: [[TMP31:%.*]] = add i64 [[INDEX]], 4
+; I32-NEXT: [[TMP32:%.*]] = add i64 [[INDEX]], 5
+; I32-NEXT: [[TMP33:%.*]] = add i64 [[INDEX]], 6
+; I32-NEXT: [[TMP34:%.*]] = add i64 [[INDEX]], 7
; I32-NEXT: [[TMP3:%.*]] = mul i64 [[INDEX]], [[OFFSET]]
; I32-NEXT: [[TMP4:%.*]] = mul i64 [[TMP0]], [[OFFSET]]
; I32-NEXT: [[TMP5:%.*]] = mul i64 [[TMP1]], [[OFFSET]]
; I32-NEXT: [[TMP6:%.*]] = mul i64 [[TMP2]], [[OFFSET]]
+; I32-NEXT: [[TMP47:%.*]] = mul i64 [[TMP31]], [[OFFSET]]
+; I32-NEXT: [[TMP48:%.*]] = mul i64 [[TMP32]], [[OFFSET]]
+; I32-NEXT: [[TMP49:%.*]] = mul i64 [[TMP33]], [[OFFSET]]
+; I32-NEXT: [[TMP50:%.*]] = mul i64 [[TMP34]], [[OFFSET]]
; I32-NEXT: [[TMP7:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP3]]
; I32-NEXT: [[TMP8:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP4]]
; I32-NEXT: [[TMP9:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP5]]
; I32-NEXT: [[TMP10:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP6]]
+; I32-NEXT: [[TMP55:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP47]]
+; I32-NEXT: [[TMP56:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP48]]
+; I32-NEXT: [[TMP57:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP49]]
+; I32-NEXT: [[TMP58:%.*]] = getelementptr i32, ptr [[SRC_0]], i64 [[TMP50]]
; I32-NEXT: [[TMP11:%.*]] = load i32, ptr [[TMP7]], align 4
; I32-NEXT: [[TMP12:%.*]] = load i32, ptr [[TMP8]], align 4
; I32-NEXT: [[TMP13:%.*]] = load i32, ptr [[TMP9]], align 4
; I32-NEXT: [[TMP14:%.*]] = load i32, ptr [[TMP10]], align 4
+; I32-NEXT: [[TMP72:%.*]] = load i32, ptr [[TMP55]], align 4
+; I32-NEXT: [[TMP73:%.*]] = load i32, ptr [[TMP56]], align 4
+; I32-NEXT: [[TMP74:%.*]] = load i32, ptr [[TMP57]], align 4
+; I32-NEXT: [[TMP75:%.*]] = load i32, ptr [[TMP58]], align 4
; I32-NEXT: [[TMP15:%.*]] = sext i32 [[TMP11]] to i64
; I32-NEXT: [[TMP16:%.*]] = sext i32 [[TMP12]] to i64
; I32-NEXT: [[TMP17:%.*]] = sext i32 [[TMP13]] to i64
; I32-NEXT: [[TMP18:%.*]] = sext i32 [[TMP14]] to i64
+; I32-NEXT: [[TMP35:%.*]] = sext i32 [[TMP72]] to i64
+; I32-NEXT: [[TMP80:%.*]] = sext i32 [[TMP73]] to i64
+; I32-NEXT: [[TMP81:%.*]] = sext i32 [[TMP74]] to i64
+; I32-NEXT: [[TMP82:%.*]] = sext i32 [[TMP75]] to i64
; I32-NEXT: [[TMP19:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP15]]
; I32-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP16]]
; I32-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP17]]
; I32-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP18]]
+; I32-NEXT: [[TMP83:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP35]]
+; I32-NEXT: [[TMP44:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP80]]
+; I32-NEXT: [[TMP45:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP81]]
+; I32-NEXT: [[TMP46:%.*]] = getelementptr double, ptr [[SRC_1]], i64 [[TMP82]]
; I32-NEXT: [[TMP23:%.*]] = getelementptr i8, ptr [[TMP19]], i64 -8
; I32-NEXT: [[TMP24:%.*]] = getelementptr i8, ptr [[TMP20]], i64 -8
; I32-NEXT: [[TMP25:%.*]] = getelementptr i8, ptr [[TMP21]], i64 -8
; I32-NEXT: [[TMP26:%.*]] = getelementptr i8, ptr [[TMP22]], i64 -8
+; I32-NEXT: [[TMP51:%.*]] = getelementptr i8, ptr [[TMP83]], i64 -8
+; I32-NEXT: [[TMP52:%.*]] = getelementptr i8, ptr [[TMP44]], i64 -8
+; I32-NEXT: [[TMP53:%.*]] = getelementptr i8, ptr [[TMP45]], i64 -8
+; I32-NEXT: [[TMP54:%.*]] = getelementptr i8, ptr [[TMP46]], i64 -8
; I32-NEXT: [[TMP27:%.*]] = load double, ptr [[TMP23]], align 8
; I32-NEXT: [[TMP28:%.*]] = load double, ptr [[TMP24]], align 8
; I32-NEXT: [[TMP29:%.*]] = load double, ptr [[TMP25]], align 8
; I32-NEXT: [[TMP30:%.*]] = load double, ptr [[TMP26]], align 8
-; I32-NEXT: [[TMP31:%.*]] = insertelement <4 x double> poison, double [[TMP27]], i64 0
-; I32-NEXT: [[TMP32:%.*]] = insertelement <4 x double> [[TMP31]], double [[TMP28]], i64 1
-; I32-NEXT: [[TMP33:%.*]] = insertelement <4 x double> [[TMP32]], double [[TMP29]], i64 2
-; I32-NEXT: [[TMP34:%.*]] = insertelement <4 x double> [[TMP33]], double [[TMP30]], i64 3
-; I32-NEXT: [[TMP35:%.*]] = fsub <4 x double> zeroinitializer, [[TMP34]]
+; I32-NEXT: [[TMP59:%.*]] = load double, ptr [[TMP51]], align 8
+; I32-NEXT: [[TMP60:%.*]] = load double, ptr [[TMP52]], align 8
+; I32-NEXT: [[TMP61:%.*]] = load double, ptr [[TMP53]], align 8
+; I32-NEXT: [[TMP62:%.*]] = load double, ptr [[TMP54]], align 8
+; I32-NEXT: [[TMP63:%.*]] = insertelement <8 x double> poison, double [[TMP27]], i64 0
+; I32-NEXT: [[TMP64:%.*]] = insertelement <8 x double> [[TMP63]], double [[TMP28]], i64 1
+; I32-NEXT: [[TMP65:%.*]] = insertelement <8 x double> [[TMP64]], double [[TMP29]], i64 2
+; I32-NEXT: [[TMP66:%.*]] = insertelement <8 x double> [[TMP65]], double [[TMP30]], i64 3
+; I32-NEXT: [[TMP67:%.*]] = insertelement <8 x double> [[TMP66]], double [[TMP59]], i64 4
+; I32-NEXT: [[TMP68:%.*]] = insertelement <8 x double> [[TMP67]], double [[TMP60]], i64 5
+; I32-NEXT: [[TMP69:%.*]] = insertelement <8 x double> [[TMP68]], double [[TMP61]], i64 6
+; I32-NEXT: [[TMP70:%.*]] = insertelement <8 x double> [[TMP69]], double [[TMP62]], i64 7
+; I32-NEXT: [[TMP71:%.*]] = fsub <8 x double> zeroinitializer, [[TMP70]]
; I32-NEXT: [[TMP40:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP3]]
; I32-NEXT: [[TMP41:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP4]]
; I32-NEXT: [[TMP42:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP5]]
; I32-NEXT: [[TMP43:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP6]]
-; I32-NEXT: [[TMP36:%.*]] = extractelement <4 x double> [[TMP35]], i64 0
+; I32-NEXT: [[TMP76:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP47]]
+; I32-NEXT: [[TMP77:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP48]]
+; I32-NEXT: [[TMP78:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP49]]
+; I32-NEXT: [[TMP79:%.*]] = getelementptr double, ptr [[DST]], i64 [[TMP50]]
+; I32-NEXT: [[TMP36:%.*]] = extractelement <8 x double> [[TMP71]], i64 0
; I32-NEXT: store double [[TMP36]], ptr [[TMP40]], align 8
-; I32-NEXT: [[TMP37:%.*]] = extractelement <4 x double> [[TMP35]], i64 1
+; I32-NEXT: [[TMP37:%.*]] = extractelement <8 x double> [[TMP71]], i64 1
; I32-NEXT: store double [[TMP37]], ptr [[TMP41]], align 8
-; I32-NEXT: [[TMP38:%.*]] = extractelement <4 x double> [[TMP35]], i64 2
+; I32-NEXT: [[TMP38:%.*]] = extractelement <8 x double> [[TMP71]], i64 2
; I32-NEXT: store double [[TMP38]], ptr [[TMP42]], align 8
-; I32-NEXT: [[TMP39:%.*]] = extractelement <4 x double> [[TMP35]], i64 3
+; I32-NEXT: [[TMP39:%.*]] = extractelement <8 x double> [[TMP71]], i64 3
; I32-NEXT: store double [[TMP39]], ptr [[TMP43]], align 8
-; I32-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
-; I32-NEXT: [[TMP44:%.*]] = icmp eq i64 [[INDEX_NEXT]], 100
-; I32-NEXT: br i1 [[TMP44]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; I32-NEXT: [[TMP84:%.*]] = extractelement <8 x double> [[TMP71]], i64 4
+; I32-NEXT: store double [[TMP84]], ptr [[TMP76]], align 8
+; I32-NEXT: [[TMP85:%.*]] = extractelement <8 x double> [[TMP71]], i64 5
+; I32-NEXT: store double [[TMP85]], ptr [[TMP77]], align 8
+; I32-NEXT: [[TMP86:%.*]] = extractelement <8 x double> [[TMP71]], i64 6
+; I32-NEXT: store double [[TMP86]], ptr [[TMP78]], align 8
+; I32-NEXT: [[TMP87:%.*]] = extractelement <8 x double> [[TMP71]], i64 7
+; I32-NEXT: store double [[TMP87]], ptr [[TMP79]], align 8
+; I32-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; I32-NEXT: [[TMP88:%.*]] = icmp eq i64 [[INDEX_NEXT]], 96
+; I32-NEXT: br i1 [[TMP88]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
; I32: [[MIDDLE_BLOCK]]:
; I32-NEXT: br label %[[SCALAR_PH:.*]]
; I32: [[SCALAR_PH]]:
@@ -1355,7 +1360,7 @@ define void @invariant_pred_store_sunk_out_of_loop(ptr noalias %dst, ptr noalias
; I32-NEXT: [[TMP5]] = add <2 x i64> [[TMP3]], splat (i64 1)
; I32-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; I32-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1000
-; I32-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; I32-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
; I32: [[MIDDLE_BLOCK]]:
; I32-NEXT: [[BIN_RDX:%.*]] = add <2 x i64> [[TMP5]], [[TMP4]]
; I32-NEXT: [[TMP7:%.*]] = call i64 @llvm.vector.reduce.add.v2i64(<2 x i64> [[BIN_RDX]])
diff --git a/llvm/test/Transforms/LoopVectorize/X86/strided_load_cost.ll b/llvm/test/Transforms/LoopVectorize/X86/strided_load_cost.ll
index 576a27bce5df4..4c5bf44ee7911 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/strided_load_cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/strided_load_cost.ll
@@ -509,12 +509,36 @@ define void @test(ptr %A, ptr noalias %B) #0 {
; CHECK-NEXT: [[TMP5:%.*]] = add i64 [[OFFSET_IDX]], 10
; CHECK-NEXT: [[TMP6:%.*]] = add i64 [[OFFSET_IDX]], 12
; CHECK-NEXT: [[TMP7:%.*]] = add i64 [[OFFSET_IDX]], 14
+; CHECK-NEXT: [[TMP8:%.*]] = add i64 [[OFFSET_IDX]], 16
+; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[OFFSET_IDX]], 18
+; CHECK-NEXT: [[TMP10:%.*]] = add i64 [[OFFSET_IDX]], 20
+; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[OFFSET_IDX]], 22
+; CHECK-NEXT: [[TMP12:%.*]] = add i64 [[OFFSET_IDX]], 24
+; CHECK-NEXT: [[TMP13:%.*]] = add i64 [[OFFSET_IDX]], 26
+; CHECK-NEXT: [[TMP14:%.*]] = add i64 [[OFFSET_IDX]], 28
+; CHECK-NEXT: [[TMP15:%.*]] = add i64 [[OFFSET_IDX]], 30
+; CHECK-NEXT: [[TMP37:%.*]] = add i64 [[OFFSET_IDX]], 32
+; CHECK-NEXT: [[TMP17:%.*]] = add i64 [[OFFSET_IDX]], 34
+; CHECK-NEXT: [[TMP18:%.*]] = add i64 [[OFFSET_IDX]], 36
+; CHECK-NEXT: [[TMP19:%.*]] = add i64 [[OFFSET_IDX]], 38
+; CHECK-NEXT: [[TMP38:%.*]] = add i64 [[OFFSET_IDX]], 40
+; CHECK-NEXT: [[TMP39:%.*]] = add i64 [[OFFSET_IDX]], 42
+; CHECK-NEXT: [[TMP40:%.*]] = add i64 [[OFFSET_IDX]], 44
+; CHECK-NEXT: [[TMP41:%.*]] = add i64 [[OFFSET_IDX]], 46
+; CHECK-NEXT: [[TMP42:%.*]] = add i64 [[OFFSET_IDX]], 48
+; CHECK-NEXT: [[TMP67:%.*]] = add i64 [[OFFSET_IDX]], 50
+; CHECK-NEXT: [[TMP68:%.*]] = add i64 [[OFFSET_IDX]], 52
+; CHECK-NEXT: [[TMP69:%.*]] = add i64 [[OFFSET_IDX]], 54
+; CHECK-NEXT: [[TMP70:%.*]] = add i64 [[OFFSET_IDX]], 56
+; CHECK-NEXT: [[TMP71:%.*]] = add i64 [[OFFSET_IDX]], 58
+; CHECK-NEXT: [[TMP72:%.*]] = add i64 [[OFFSET_IDX]], 60
+; CHECK-NEXT: [[TMP73:%.*]] = add i64 [[OFFSET_IDX]], 62
; CHECK-NEXT: [[TMP16:%.*]] = getelementptr inbounds [1024 x i32], ptr [[A]], i64 0, i64 [[OFFSET_IDX]]
-; CHECK-NEXT: [[WIDE_VEC:%.*]] = load <16 x i32>, ptr [[TMP16]], align 4
-; CHECK-NEXT: [[STRIDED_VEC:%.*]] = shufflevector <16 x i32> [[WIDE_VEC]], <16 x i32> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
-; CHECK-NEXT: [[STRIDED_VEC1:%.*]] = shufflevector <16 x i32> [[WIDE_VEC]], <16 x i32> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
-; CHECK-NEXT: [[TMP18:%.*]] = add <8 x i32> [[STRIDED_VEC]], [[STRIDED_VEC1]]
-; CHECK-NEXT: [[TMP19:%.*]] = trunc <8 x i32> [[TMP18]] to <8 x i8>
+; CHECK-NEXT: [[WIDE_VEC:%.*]] = load <64 x i32>, ptr [[TMP16]], align 4
+; CHECK-NEXT: [[STRIDED_VEC:%.*]] = shufflevector <64 x i32> [[WIDE_VEC]], <64 x i32> poison, <32 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30, i32 32, i32 34, i32 36, i32 38, i32 40, i32 42, i32 44, i32 46, i32 48, i32 50, i32 52, i32 54, i32 56, i32 58, i32 60, i32 62>
+; CHECK-NEXT: [[STRIDED_VEC1:%.*]] = shufflevector <64 x i32> [[WIDE_VEC]], <64 x i32> poison, <32 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31, i32 33, i32 35, i32 37, i32 39, i32 41, i32 43, i32 45, i32 47, i32 49, i32 51, i32 53, i32 55, i32 57, i32 59, i32 61, i32 63>
+; CHECK-NEXT: [[TMP74:%.*]] = add <32 x i32> [[STRIDED_VEC]], [[STRIDED_VEC1]]
+; CHECK-NEXT: [[TMP99:%.*]] = trunc <32 x i32> [[TMP74]] to <32 x i8>
; CHECK-NEXT: [[TMP20:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[OFFSET_IDX]]
; CHECK-NEXT: [[TMP21:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP1]]
; CHECK-NEXT: [[TMP22:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP2]]
@@ -523,23 +547,95 @@ define void @test(ptr %A, ptr noalias %B) #0 {
; CHECK-NEXT: [[TMP25:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP5]]
; CHECK-NEXT: [[TMP26:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP6]]
; CHECK-NEXT: [[TMP27:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP7]]
-; CHECK-NEXT: [[TMP28:%.*]] = extractelement <8 x i8> [[TMP19]], i64 0
+; CHECK-NEXT: [[TMP43:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP8]]
+; CHECK-NEXT: [[TMP44:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP9]]
+; CHECK-NEXT: [[TMP45:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP10]]
+; CHECK-NEXT: [[TMP46:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP11]]
+; CHECK-NEXT: [[TMP47:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP12]]
+; CHECK-NEXT: [[TMP48:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP13]]
+; CHECK-NEXT: [[TMP49:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP14]]
+; CHECK-NEXT: [[TMP50:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP15]]
+; CHECK-NEXT: [[TMP51:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP37]]
+; CHECK-NEXT: [[TMP52:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP17]]
+; CHECK-NEXT: [[TMP53:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP18]]
+; CHECK-NEXT: [[TMP54:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP19]]
+; CHECK-NEXT: [[TMP55:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP38]]
+; CHECK-NEXT: [[TMP56:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP39]]
+; CHECK-NEXT: [[TMP57:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP40]]
+; CHECK-NEXT: [[TMP58:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP41]]
+; CHECK-NEXT: [[TMP59:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP42]]
+; CHECK-NEXT: [[TMP60:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP67]]
+; CHECK-NEXT: [[TMP61:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP68]]
+; CHECK-NEXT: [[TMP62:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP69]]
+; CHECK-NEXT: [[TMP63:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP70]]
+; CHECK-NEXT: [[TMP64:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP71]]
+; CHECK-NEXT: [[TMP65:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP72]]
+; CHECK-NEXT: [[TMP66:%.*]] = getelementptr inbounds [1024 x i8], ptr [[B]], i64 0, i64 [[TMP73]]
+; CHECK-NEXT: [[TMP28:%.*]] = extractelement <32 x i8> [[TMP99]], i64 0
; CHECK-NEXT: store i8 [[TMP28]], ptr [[TMP20]], align 1
-; CHECK-NEXT: [[TMP29:%.*]] = extractelement <8 x i8> [[TMP19]], i64 1
+; CHECK-NEXT: [[TMP29:%.*]] = extractelement <32 x i8> [[TMP99]], i64 1
; CHECK-NEXT: store i8 [[TMP29]], ptr [[TMP21]], align 1
-; CHECK-NEXT: [[TMP30:%.*]] = extractelement <8 x i8> [[TMP19]], i64 2
+; CHECK-NEXT: [[TMP30:%.*]] = extractelement <32 x i8> [[TMP99]], i64 2
; CHECK-NEXT: store i8 [[TMP30]], ptr [[TMP22]], align 1
-; CHECK-NEXT: [[TMP31:%.*]] = extractelement <8 x i8> [[TMP19]], i64 3
+; CHECK-NEXT: [[TMP31:%.*]] = extractelement <32 x i8> [[TMP99]], i64 3
; CHECK-NEXT: store i8 [[TMP31]], ptr [[TMP23]], align 1
-; CHECK-NEXT: [[TMP32:%.*]] = extractelement <8 x i8> [[TMP19]], i64 4
+; CHECK-NEXT: [[TMP32:%.*]] = extractelement <32 x i8> [[TMP99]], i64 4
; CHECK-NEXT: store i8 [[TMP32]], ptr [[TMP24]], align 1
-; CHECK-NEXT: [[TMP33:%.*]] = extractelement <8 x i8> [[TMP19]], i64 5
+; CHECK-NEXT: [[TMP33:%.*]] = extractelement <32 x i8> [[TMP99]], i64 5
; CHECK-NEXT: store i8 [[TMP33]], ptr [[TMP25]], align 1
-; CHECK-NEXT: [[TMP34:%.*]] = extractelement <8 x i8> [[TMP19]], i64 6
+; CHECK-NEXT: [[TMP34:%.*]] = extractelement <32 x i8> [[TMP99]], i64 6
; CHECK-NEXT: store i8 [[TMP34]], ptr [[TMP26]], align 1
-; CHECK-NEXT: [[TMP35:%.*]] = extractelement <8 x i8> [[TMP19]], i64 7
+; CHECK-NEXT: [[TMP35:%.*]] = extractelement <32 x i8> [[TMP99]], i64 7
; CHECK-NEXT: store i8 [[TMP35]], ptr [[TMP27]], align 1
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-NEXT: [[TMP75:%.*]] = extractelement <32 x i8> [[TMP99]], i64 8
+; CHECK-NEXT: store i8 [[TMP75]], ptr [[TMP43]], align 1
+; CHECK-NEXT: [[TMP76:%.*]] = extractelement <32 x i8> [[TMP99]], i64 9
+; CHECK-NEXT: store i8 [[TMP76]], ptr [[TMP44]], align 1
+; CHECK-NEXT: [[TMP77:%.*]] = extractelement <32 x i8> [[TMP99]], i64 10
+; CHECK-NEXT: store i8 [[TMP77]], ptr [[TMP45]], align 1
+; CHECK-NEXT: [[TMP78:%.*]] = extractelement <32 x i8> [[TMP99]], i64 11
+; CHECK-NEXT: store i8 [[TMP78]], ptr [[TMP46]], align 1
+; CHECK-NEXT: [[TMP79:%.*]] = extractelement <32 x i8> [[TMP99]], i64 12
+; CHECK-NEXT: store i8 [[TMP79]], ptr [[TMP47]], align 1
+; CHECK-NEXT: [[TMP80:%.*]] = extractelement <32 x i8> [[TMP99]], i64 13
+; CHECK-NEXT: store i8 [[TMP80]], ptr [[TMP48]], align 1
+; CHECK-NEXT: [[TMP81:%.*]] = extractelement <32 x i8> [[TMP99]], i64 14
+; CHECK-NEXT: store i8 [[TMP81]], ptr [[TMP49]], align 1
+; CHECK-NEXT: [[TMP82:%.*]] = extractelement <32 x i8> [[TMP99]], i64 15
+; CHECK-NEXT: store i8 [[TMP82]], ptr [[TMP50]], align 1
+; CHECK-NEXT: [[TMP83:%.*]] = extractelement <32 x i8> [[TMP99]], i64 16
+; CHECK-NEXT: store i8 [[TMP83]], ptr [[TMP51]], align 1
+; CHECK-NEXT: [[TMP84:%.*]] = extractelement <32 x i8> [[TMP99]], i64 17
+; CHECK-NEXT: store i8 [[TMP84]], ptr [[TMP52]], align 1
+; CHECK-NEXT: [[TMP85:%.*]] = extractelement <32 x i8> [[TMP99]], i64 18
+; CHECK-NEXT: store i8 [[TMP85]], ptr [[TMP53]], align 1
+; CHECK-NEXT: [[TMP86:%.*]] = extractelement <32 x i8> [[TMP99]], i64 19
+; CHECK-NEXT: store i8 [[TMP86]], ptr [[TMP54]], align 1
+; CHECK-NEXT: [[TMP87:%.*]] = extractelement <32 x i8> [[TMP99]], i64 20
+; CHECK-NEXT: store i8 [[TMP87]], ptr [[TMP55]], align 1
+; CHECK-NEXT: [[TMP88:%.*]] = extractelement <32 x i8> [[TMP99]], i64 21
+; CHECK-NEXT: store i8 [[TMP88]], ptr [[TMP56]], align 1
+; CHECK-NEXT: [[TMP89:%.*]] = extractelement <32 x i8> [[TMP99]], i64 22
+; CHECK-NEXT: store i8 [[TMP89]], ptr [[TMP57]], align 1
+; CHECK-NEXT: [[TMP90:%.*]] = extractelement <32 x i8> [[TMP99]], i64 23
+; CHECK-NEXT: store i8 [[TMP90]], ptr [[TMP58]], align 1
+; CHECK-NEXT: [[TMP91:%.*]] = extractelement <32 x i8> [[TMP99]], i64 24
+; CHECK-NEXT: store i8 [[TMP91]], ptr [[TMP59]], align 1
+; CHECK-NEXT: [[TMP92:%.*]] = extractelement <32 x i8> [[TMP99]], i64 25
+; CHECK-NEXT: store i8 [[TMP92]], ptr [[TMP60]], align 1
+; CHECK-NEXT: [[TMP93:%.*]] = extractelement <32 x i8> [[TMP99]], i64 26
+; CHECK-NEXT: store i8 [[TMP93]], ptr [[TMP61]], align 1
+; CHECK-NEXT: [[TMP94:%.*]] = extractelement <32 x i8> [[TMP99]], i64 27
+; CHECK-NEXT: store i8 [[TMP94]], ptr [[TMP62]], align 1
+; CHECK-NEXT: [[TMP95:%.*]] = extractelement <32 x i8> [[TMP99]], i64 28
+; CHECK-NEXT: store i8 [[TMP95]], ptr [[TMP63]], align 1
+; CHECK-NEXT: [[TMP96:%.*]] = extractelement <32 x i8> [[TMP99]], i64 29
+; CHECK-NEXT: store i8 [[TMP96]], ptr [[TMP64]], align 1
+; CHECK-NEXT: [[TMP97:%.*]] = extractelement <32 x i8> [[TMP99]], i64 30
+; CHECK-NEXT: store i8 [[TMP97]], ptr [[TMP65]], align 1
+; CHECK-NEXT: [[TMP98:%.*]] = extractelement <32 x i8> [[TMP99]], i64 31
+; CHECK-NEXT: store i8 [[TMP98]], ptr [[TMP66]], align 1
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; CHECK-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT]], 512
; CHECK-NEXT: br i1 [[TMP36]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/vector_ptr_load_store.ll b/llvm/test/Transforms/LoopVectorize/X86/vector_ptr_load_store.ll
index 91907c3e4d69e..fe886aac6f9b5 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/vector_ptr_load_store.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/vector_ptr_load_store.ll
@@ -96,7 +96,7 @@ define void @test_nonconsecutive_store() {
;; pointer types into account.
; CHECK: test_consecutive_ptr_load
; CHECK: LV: The Smallest and Widest types: 8 / 64 bits.
-; CHECK: LV: Selecting VF: 4
+; CHECK: LV: Selecting VF: 16
define i8 @test_consecutive_ptr_load() readonly {
br label %1
@@ -121,7 +121,7 @@ define i8 @test_consecutive_ptr_load() readonly {
;; However, we should not take unconsecutive loads of pointers into account.
; CHECK: test_nonconsecutive_ptr_load
; CHECK: LV: The Smallest and Widest types: 16 / 64 bits.
-; CHECK: LV: Selecting VF: 1
+; CHECK: LV: Selecting VF: 16
define void @test_nonconsecutive_ptr_load() {
br label %1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks-loopid-dbg.ll b/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks-loopid-dbg.ll
index 9ded7fa7d6a3c..ef6a4b37878f8 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks-loopid-dbg.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks-loopid-dbg.ll
@@ -1,14 +1,133 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
; RUN: opt < %s -passes=loop-vectorize -mtriple=x86_64-unknown-linux -S -pass-remarks='loop-vectorize' 2>&1 | FileCheck -check-prefix=VECTORIZED %s
; RUN: opt < %s -passes=loop-vectorize -force-vector-width=1 -force-vector-interleave=4 -mtriple=x86_64-unknown-linux -S -pass-remarks='loop-vectorize' 2>&1 | FileCheck -check-prefix=UNROLLED %s
; RUN: opt < %s -passes=loop-vectorize -force-vector-width=1 -force-vector-interleave=1 -mtriple=x86_64-unknown-linux -S -pass-remarks-analysis='loop-vectorize' 2>&1 | FileCheck -check-prefix=NONE %s
-; VECTORIZED: remark: vectorization-remarks.c:17:8: vectorized loop (vectorization width: 4, interleaved count: 2)
+; VECTORIZED: remark: vectorization-remarks.c:17:8: vectorized loop (vectorization width: 16, interleaved count: 1)
; UNROLLED: remark: vectorization-remarks.c:17:8: interleaved loop (interleaved count: 4)
; NONE: remark: vectorization-remarks.c:17:8: loop not vectorized: vectorization and interleaving are explicitly disabled, or the loop has already been vectorized
target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
define i32 @foo(i32 %n) #0 !dbg !4 {
+; VECTORIZED-LABEL: define i32 @foo(
+; VECTORIZED-SAME: i32 [[N:%.*]]) !dbg [[DBG4:![0-9]+]] {
+; VECTORIZED-NEXT: [[ENTRY:.*:]]
+; VECTORIZED-NEXT: [[DIFF:%.*]] = alloca i32, align 4
+; VECTORIZED-NEXT: [[CB:%.*]] = alloca [16 x i8], align 16
+; VECTORIZED-NEXT: [[CC:%.*]] = alloca [16 x i8], align 16
+; VECTORIZED-NEXT: store i32 0, ptr [[DIFF]], align 4
+; VECTORIZED-NEXT: br label %[[VECTOR_PH:.*]]
+; VECTORIZED: [[VECTOR_PH]]:
+; VECTORIZED-NEXT: br label %[[VECTOR_BODY:.*]]
+; VECTORIZED: [[VECTOR_BODY]]:
+; VECTORIZED-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[CB]], align 1
+; VECTORIZED-NEXT: [[TMP0:%.*]] = sext <16 x i8> [[WIDE_LOAD]] to <16 x i32>
+; VECTORIZED-NEXT: [[WIDE_LOAD1:%.*]] = load <16 x i8>, ptr [[CC]], align 1
+; VECTORIZED-NEXT: [[TMP1:%.*]] = sext <16 x i8> [[WIDE_LOAD1]] to <16 x i32>
+; VECTORIZED-NEXT: [[TMP2:%.*]] = sub <16 x i32> [[TMP0]], [[TMP1]]
+; VECTORIZED-NEXT: [[TMP3:%.*]] = add <16 x i32> [[TMP2]], zeroinitializer
+; VECTORIZED-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; VECTORIZED: [[MIDDLE_BLOCK]]:
+; VECTORIZED-NEXT: [[TMP4:%.*]] = call i32 @llvm.vector.reduce.add.v16i32(<16 x i32> [[TMP3]])
+; VECTORIZED-NEXT: br label %[[FOR_END:.*]]
+; VECTORIZED: [[FOR_END]]:
+; VECTORIZED-NEXT: store i32 [[TMP4]], ptr [[DIFF]], align 4
+; VECTORIZED-NEXT: call void @ibar(ptr [[DIFF]])
+; VECTORIZED-NEXT: ret i32 0
+;
+; UNROLLED-LABEL: define i32 @foo(
+; UNROLLED-SAME: i32 [[N:%.*]]) !dbg [[DBG4:![0-9]+]] {
+; UNROLLED-NEXT: [[ENTRY:.*:]]
+; UNROLLED-NEXT: [[DIFF:%.*]] = alloca i32, align 4
+; UNROLLED-NEXT: [[CB:%.*]] = alloca [16 x i8], align 16
+; UNROLLED-NEXT: [[CC:%.*]] = alloca [16 x i8], align 16
+; UNROLLED-NEXT: store i32 0, ptr [[DIFF]], align 4
+; UNROLLED-NEXT: br label %[[VECTOR_PH:.*]]
+; UNROLLED: [[VECTOR_PH]]:
+; UNROLLED-NEXT: br label %[[VECTOR_BODY:.*]]
+; UNROLLED: [[VECTOR_BODY]]:
+; UNROLLED-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[VEC_PHI:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP31:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[VEC_PHI1:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP32:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[VEC_PHI2:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP33:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[VEC_PHI3:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP34:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[TMP0:%.*]] = add i64 [[INDEX]], 1
+; UNROLLED-NEXT: [[TMP1:%.*]] = add i64 [[INDEX]], 2
+; UNROLLED-NEXT: [[TMP2:%.*]] = add i64 [[INDEX]], 3
+; UNROLLED-NEXT: [[TMP3:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[INDEX]]
+; UNROLLED-NEXT: [[TMP4:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[TMP0]]
+; UNROLLED-NEXT: [[TMP5:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[TMP1]]
+; UNROLLED-NEXT: [[TMP6:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[TMP2]]
+; UNROLLED-NEXT: [[TMP7:%.*]] = load i8, ptr [[TMP3]], align 1
+; UNROLLED-NEXT: [[TMP8:%.*]] = load i8, ptr [[TMP4]], align 1
+; UNROLLED-NEXT: [[TMP9:%.*]] = load i8, ptr [[TMP5]], align 1
+; UNROLLED-NEXT: [[TMP10:%.*]] = load i8, ptr [[TMP6]], align 1
+; UNROLLED-NEXT: [[TMP11:%.*]] = sext i8 [[TMP7]] to i32
+; UNROLLED-NEXT: [[TMP12:%.*]] = sext i8 [[TMP8]] to i32
+; UNROLLED-NEXT: [[TMP13:%.*]] = sext i8 [[TMP9]] to i32
+; UNROLLED-NEXT: [[TMP14:%.*]] = sext i8 [[TMP10]] to i32
+; UNROLLED-NEXT: [[TMP15:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[INDEX]]
+; UNROLLED-NEXT: [[TMP16:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[TMP0]]
+; UNROLLED-NEXT: [[TMP17:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[TMP1]]
+; UNROLLED-NEXT: [[TMP18:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[TMP2]]
+; UNROLLED-NEXT: [[TMP19:%.*]] = load i8, ptr [[TMP15]], align 1
+; UNROLLED-NEXT: [[TMP20:%.*]] = load i8, ptr [[TMP16]], align 1
+; UNROLLED-NEXT: [[TMP21:%.*]] = load i8, ptr [[TMP17]], align 1
+; UNROLLED-NEXT: [[TMP22:%.*]] = load i8, ptr [[TMP18]], align 1
+; UNROLLED-NEXT: [[TMP23:%.*]] = sext i8 [[TMP19]] to i32
+; UNROLLED-NEXT: [[TMP24:%.*]] = sext i8 [[TMP20]] to i32
+; UNROLLED-NEXT: [[TMP25:%.*]] = sext i8 [[TMP21]] to i32
+; UNROLLED-NEXT: [[TMP26:%.*]] = sext i8 [[TMP22]] to i32
+; UNROLLED-NEXT: [[TMP27:%.*]] = sub i32 [[TMP11]], [[TMP23]]
+; UNROLLED-NEXT: [[TMP28:%.*]] = sub i32 [[TMP12]], [[TMP24]]
+; UNROLLED-NEXT: [[TMP29:%.*]] = sub i32 [[TMP13]], [[TMP25]]
+; UNROLLED-NEXT: [[TMP30:%.*]] = sub i32 [[TMP14]], [[TMP26]]
+; UNROLLED-NEXT: [[TMP31]] = add i32 [[TMP27]], [[VEC_PHI]]
+; UNROLLED-NEXT: [[TMP32]] = add i32 [[TMP28]], [[VEC_PHI1]]
+; UNROLLED-NEXT: [[TMP33]] = add i32 [[TMP29]], [[VEC_PHI2]]
+; UNROLLED-NEXT: [[TMP34]] = add i32 [[TMP30]], [[VEC_PHI3]]
+; UNROLLED-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; UNROLLED-NEXT: [[TMP35:%.*]] = icmp eq i64 [[INDEX_NEXT]], 16
+; UNROLLED-NEXT: br i1 [[TMP35]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; UNROLLED: [[MIDDLE_BLOCK]]:
+; UNROLLED-NEXT: [[BIN_RDX:%.*]] = add i32 [[TMP32]], [[TMP31]]
+; UNROLLED-NEXT: [[BIN_RDX4:%.*]] = add i32 [[TMP33]], [[BIN_RDX]]
+; UNROLLED-NEXT: [[BIN_RDX5:%.*]] = add i32 [[TMP34]], [[BIN_RDX4]]
+; UNROLLED-NEXT: br label %[[FOR_END:.*]]
+; UNROLLED: [[FOR_END]]:
+; UNROLLED-NEXT: store i32 [[BIN_RDX5]], ptr [[DIFF]], align 4
+; UNROLLED-NEXT: call void @ibar(ptr [[DIFF]])
+; UNROLLED-NEXT: ret i32 0
+;
+; NONE-LABEL: define i32 @foo(
+; NONE-SAME: i32 [[N:%.*]]) !dbg [[DBG4:![0-9]+]] {
+; NONE-NEXT: [[ENTRY:.*]]:
+; NONE-NEXT: [[DIFF:%.*]] = alloca i32, align 4
+; NONE-NEXT: [[CB:%.*]] = alloca [16 x i8], align 16
+; NONE-NEXT: [[CC:%.*]] = alloca [16 x i8], align 16
+; NONE-NEXT: store i32 0, ptr [[DIFF]], align 4
+; NONE-NEXT: br label %[[FOR_BODY:.*]]
+; NONE: [[FOR_BODY]]:
+; NONE-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDVARS_IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; NONE-NEXT: [[ADD8:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ADD:%.*]], %[[FOR_BODY]] ]
+; NONE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[INDVARS_IV]]
+; NONE-NEXT: [[TMP0:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
+; NONE-NEXT: [[CONV:%.*]] = sext i8 [[TMP0]] to i32
+; NONE-NEXT: [[ARRAYIDX2:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[INDVARS_IV]]
+; NONE-NEXT: [[TMP1:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
+; NONE-NEXT: [[CONV3:%.*]] = sext i8 [[TMP1]] to i32
+; NONE-NEXT: [[SUB:%.*]] = sub i32 [[CONV]], [[CONV3]]
+; NONE-NEXT: [[ADD]] = add nsw i32 [[SUB]], [[ADD8]]
+; NONE-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1
+; NONE-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], 16
+; NONE-NEXT: br i1 [[EXITCOND]], label %[[FOR_END:.*]], label %[[FOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; NONE: [[FOR_END]]:
+; NONE-NEXT: [[ADD_LCSSA:%.*]] = phi i32 [ [[ADD]], %[[FOR_BODY]] ]
+; NONE-NEXT: store i32 [[ADD_LCSSA]], ptr [[DIFF]], align 4
+; NONE-NEXT: call void @ibar(ptr [[DIFF]])
+; NONE-NEXT: ret i32 0
+;
entry:
%diff = alloca i32, align 4
%cb = alloca [16 x i8], align 16
@@ -61,3 +180,31 @@ declare void @ibar(ptr) #1
!23 = !DILocation(line: 21, column: 3, scope: !4)
!24 = distinct !DICompileUnit(language: DW_LANG_C89, file: !1, emissionKind: NoDebug)
!25 = !{!25, !15}
+;.
+; VECTORIZED: [[META2:![0-9]+]] = distinct !DICompileUnit(language: DW_LANG_C89, file: [[META3:![0-9]+]], isOptimized: false, runtimeVersion: 0, emissionKind: NoDebug)
+; VECTORIZED: [[META3]] = !DIFile(filename: "{{.*}}vectorization-remarks.c", directory: {{.*}})
+; VECTORIZED: [[DBG4]] = distinct !DISubprogram(name: "foo", scope: [[META3]], file: [[META3]], line: 5, type: [[META5:![0-9]+]], scopeLine: 6, virtualIndex: 6, flags: DIFlagPrototyped, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META2]], retainedNodes: [[META6:![0-9]+]])
+; VECTORIZED: [[META5]] = !DISubroutineType(types: [[META6]])
+; VECTORIZED: [[META6]] = !{}
+;.
+; UNROLLED: [[META2:![0-9]+]] = distinct !DICompileUnit(language: DW_LANG_C89, file: [[META3:![0-9]+]], isOptimized: false, runtimeVersion: 0, emissionKind: NoDebug)
+; UNROLLED: [[META3]] = !DIFile(filename: "{{.*}}vectorization-remarks.c", directory: {{.*}})
+; UNROLLED: [[DBG4]] = distinct !DISubprogram(name: "foo", scope: [[META3]], file: [[META3]], line: 5, type: [[META5:![0-9]+]], scopeLine: 6, virtualIndex: 6, flags: DIFlagPrototyped, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META2]], retainedNodes: [[META6:![0-9]+]])
+; UNROLLED: [[META5]] = !DISubroutineType(types: [[META6]])
+; UNROLLED: [[META6]] = !{}
+; UNROLLED: [[LOOP7]] = distinct !{[[LOOP7]], [[META8:![0-9]+]], [[META9:![0-9]+]], [[META10:![0-9]+]]}
+; UNROLLED: [[META8]] = !{!"llvm.loop.isvectorized", i32 1}
+; UNROLLED: [[META9]] = !{!"llvm.loop.vectorize.body", i32 1}
+; UNROLLED: [[META10]] = !{!"llvm.loop.unroll.runtime.disable"}
+;.
+; NONE: [[META2:![0-9]+]] = distinct !DICompileUnit(language: DW_LANG_C89, file: [[META3:![0-9]+]], isOptimized: false, runtimeVersion: 0, emissionKind: NoDebug)
+; NONE: [[META3]] = !DIFile(filename: "{{.*}}vectorization-remarks.c", directory: {{.*}})
+; NONE: [[DBG4]] = distinct !DISubprogram(name: "foo", scope: [[META3]], file: [[META3]], line: 5, type: [[META5:![0-9]+]], scopeLine: 6, virtualIndex: 6, flags: DIFlagPrototyped, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META2]], retainedNodes: [[META6:![0-9]+]])
+; NONE: [[META5]] = !DISubroutineType(types: [[META6]])
+; NONE: [[META6]] = !{}
+; NONE: [[LOOP7]] = distinct !{[[LOOP7]], [[META8:![0-9]+]]}
+; NONE: [[META8]] = !DILocation(line: 17, column: 8, scope: [[META9:![0-9]+]])
+; NONE: [[META9]] = distinct !DILexicalBlock(scope: [[META10:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; NONE: [[META10]] = distinct !DILexicalBlock(scope: [[META11:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; NONE: [[META11]] = distinct !DILexicalBlock(scope: [[DBG4]], file: [[META3]], line: 17, column: 3)
+;.
diff --git a/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks.ll b/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks.ll
index 41ad9ec20cc3d..398f2913bce75 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks.ll
@@ -1,14 +1,133 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
; RUN: opt < %s -passes=loop-vectorize -mtriple=x86_64-unknown-linux -S -pass-remarks='loop-vectorize' 2>&1 | FileCheck -check-prefix=VECTORIZED %s
; RUN: opt < %s -passes=loop-vectorize -force-vector-width=1 -force-vector-interleave=4 -mtriple=x86_64-unknown-linux -S -pass-remarks='loop-vectorize' 2>&1 | FileCheck -check-prefix=UNROLLED %s
; RUN: opt < %s -passes=loop-vectorize -force-vector-width=1 -force-vector-interleave=1 -mtriple=x86_64-unknown-linux -S -pass-remarks-analysis='loop-vectorize' 2>&1 | FileCheck -check-prefix=NONE %s
-; VECTORIZED: remark: vectorization-remarks.c:17:8: vectorized loop (vectorization width: 4, interleaved count: 2)
+; VECTORIZED: remark: vectorization-remarks.c:17:8: vectorized loop (vectorization width: 16, interleaved count: 1)
; UNROLLED: remark: vectorization-remarks.c:17:8: interleaved loop (interleaved count: 4)
; NONE: remark: vectorization-remarks.c:17:8: loop not vectorized: vectorization and interleaving are explicitly disabled, or the loop has already been vectorized
target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
define i32 @foo(i32 %n) #0 !dbg !4 {
+; VECTORIZED-LABEL: define i32 @foo(
+; VECTORIZED-SAME: i32 [[N:%.*]]) !dbg [[DBG4:![0-9]+]] {
+; VECTORIZED-NEXT: [[ENTRY:.*:]]
+; VECTORIZED-NEXT: [[DIFF:%.*]] = alloca i32, align 4
+; VECTORIZED-NEXT: [[CB:%.*]] = alloca [16 x i8], align 16
+; VECTORIZED-NEXT: [[CC:%.*]] = alloca [16 x i8], align 16
+; VECTORIZED-NEXT: store i32 0, ptr [[DIFF]], align 4, !dbg [[DBG7:![0-9]+]]
+; VECTORIZED-NEXT: br label %[[VECTOR_PH:.*]], !dbg [[DBG8:![0-9]+]]
+; VECTORIZED: [[VECTOR_PH]]:
+; VECTORIZED-NEXT: br label %[[VECTOR_BODY:.*]], !dbg [[DBG8]]
+; VECTORIZED: [[VECTOR_BODY]]:
+; VECTORIZED-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[CB]], align 1, !dbg [[DBG12:![0-9]+]]
+; VECTORIZED-NEXT: [[TMP0:%.*]] = sext <16 x i8> [[WIDE_LOAD]] to <16 x i32>, !dbg [[DBG12]]
+; VECTORIZED-NEXT: [[WIDE_LOAD1:%.*]] = load <16 x i8>, ptr [[CC]], align 1, !dbg [[DBG12]]
+; VECTORIZED-NEXT: [[TMP1:%.*]] = sext <16 x i8> [[WIDE_LOAD1]] to <16 x i32>, !dbg [[DBG12]]
+; VECTORIZED-NEXT: [[TMP2:%.*]] = sub <16 x i32> [[TMP0]], [[TMP1]], !dbg [[DBG12]]
+; VECTORIZED-NEXT: [[TMP3:%.*]] = add <16 x i32> [[TMP2]], zeroinitializer, !dbg [[DBG12]]
+; VECTORIZED-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; VECTORIZED: [[MIDDLE_BLOCK]]:
+; VECTORIZED-NEXT: [[TMP4:%.*]] = call i32 @llvm.vector.reduce.add.v16i32(<16 x i32> [[TMP3]]), !dbg [[DBG8]]
+; VECTORIZED-NEXT: br label %[[FOR_END:.*]], !dbg [[DBG12]]
+; VECTORIZED: [[FOR_END]]:
+; VECTORIZED-NEXT: store i32 [[TMP4]], ptr [[DIFF]], align 4, !dbg [[DBG12]]
+; VECTORIZED-NEXT: call void @ibar(ptr [[DIFF]]), !dbg [[DBG14:![0-9]+]]
+; VECTORIZED-NEXT: ret i32 0, !dbg [[DBG15:![0-9]+]]
+;
+; UNROLLED-LABEL: define i32 @foo(
+; UNROLLED-SAME: i32 [[N:%.*]]) !dbg [[DBG4:![0-9]+]] {
+; UNROLLED-NEXT: [[ENTRY:.*:]]
+; UNROLLED-NEXT: [[DIFF:%.*]] = alloca i32, align 4
+; UNROLLED-NEXT: [[CB:%.*]] = alloca [16 x i8], align 16
+; UNROLLED-NEXT: [[CC:%.*]] = alloca [16 x i8], align 16
+; UNROLLED-NEXT: store i32 0, ptr [[DIFF]], align 4, !dbg [[DBG7:![0-9]+]]
+; UNROLLED-NEXT: br label %[[VECTOR_PH:.*]], !dbg [[DBG8:![0-9]+]]
+; UNROLLED: [[VECTOR_PH]]:
+; UNROLLED-NEXT: br label %[[VECTOR_BODY:.*]], !dbg [[DBG8]]
+; UNROLLED: [[VECTOR_BODY]]:
+; UNROLLED-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ], !dbg [[DBG8]]
+; UNROLLED-NEXT: [[VEC_PHI:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP31:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[VEC_PHI1:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP32:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[VEC_PHI2:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP33:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[VEC_PHI3:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[TMP34:%.*]], %[[VECTOR_BODY]] ]
+; UNROLLED-NEXT: [[TMP0:%.*]] = add i64 [[INDEX]], 1
+; UNROLLED-NEXT: [[TMP1:%.*]] = add i64 [[INDEX]], 2
+; UNROLLED-NEXT: [[TMP2:%.*]] = add i64 [[INDEX]], 3
+; UNROLLED-NEXT: [[TMP3:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[INDEX]], !dbg [[DBG12:![0-9]+]]
+; UNROLLED-NEXT: [[TMP4:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[TMP0]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP5:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[TMP1]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP6:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[TMP2]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP7:%.*]] = load i8, ptr [[TMP3]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP8:%.*]] = load i8, ptr [[TMP4]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP9:%.*]] = load i8, ptr [[TMP5]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP10:%.*]] = load i8, ptr [[TMP6]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP11:%.*]] = sext i8 [[TMP7]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP12:%.*]] = sext i8 [[TMP8]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP13:%.*]] = sext i8 [[TMP9]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP14:%.*]] = sext i8 [[TMP10]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP15:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[INDEX]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP16:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[TMP0]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP17:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[TMP1]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP18:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[TMP2]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP19:%.*]] = load i8, ptr [[TMP15]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP20:%.*]] = load i8, ptr [[TMP16]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP21:%.*]] = load i8, ptr [[TMP17]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP22:%.*]] = load i8, ptr [[TMP18]], align 1, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP23:%.*]] = sext i8 [[TMP19]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP24:%.*]] = sext i8 [[TMP20]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP25:%.*]] = sext i8 [[TMP21]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP26:%.*]] = sext i8 [[TMP22]] to i32, !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP27:%.*]] = sub i32 [[TMP11]], [[TMP23]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP28:%.*]] = sub i32 [[TMP12]], [[TMP24]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP29:%.*]] = sub i32 [[TMP13]], [[TMP25]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP30:%.*]] = sub i32 [[TMP14]], [[TMP26]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP31]] = add i32 [[TMP27]], [[VEC_PHI]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP32]] = add i32 [[TMP28]], [[VEC_PHI1]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP33]] = add i32 [[TMP29]], [[VEC_PHI2]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[TMP34]] = add i32 [[TMP30]], [[VEC_PHI3]], !dbg [[DBG12]]
+; UNROLLED-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4, !dbg [[DBG8]]
+; UNROLLED-NEXT: [[TMP35:%.*]] = icmp eq i64 [[INDEX_NEXT]], 16, !dbg [[DBG8]]
+; UNROLLED-NEXT: br i1 [[TMP35]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !dbg [[DBG8]], !llvm.loop [[LOOP14:![0-9]+]]
+; UNROLLED: [[MIDDLE_BLOCK]]:
+; UNROLLED-NEXT: [[BIN_RDX:%.*]] = add i32 [[TMP32]], [[TMP31]], !dbg [[DBG8]]
+; UNROLLED-NEXT: [[BIN_RDX4:%.*]] = add i32 [[TMP33]], [[BIN_RDX]], !dbg [[DBG8]]
+; UNROLLED-NEXT: [[BIN_RDX5:%.*]] = add i32 [[TMP34]], [[BIN_RDX4]], !dbg [[DBG8]]
+; UNROLLED-NEXT: br label %[[FOR_END:.*]], !dbg [[DBG8]]
+; UNROLLED: [[FOR_END]]:
+; UNROLLED-NEXT: store i32 [[BIN_RDX5]], ptr [[DIFF]], align 4, !dbg [[DBG12]]
+; UNROLLED-NEXT: call void @ibar(ptr [[DIFF]]), !dbg [[DBG18:![0-9]+]]
+; UNROLLED-NEXT: ret i32 0, !dbg [[DBG19:![0-9]+]]
+;
+; NONE-LABEL: define i32 @foo(
+; NONE-SAME: i32 [[N:%.*]]) !dbg [[DBG4:![0-9]+]] {
+; NONE-NEXT: [[ENTRY:.*]]:
+; NONE-NEXT: [[DIFF:%.*]] = alloca i32, align 4
+; NONE-NEXT: [[CB:%.*]] = alloca [16 x i8], align 16
+; NONE-NEXT: [[CC:%.*]] = alloca [16 x i8], align 16
+; NONE-NEXT: store i32 0, ptr [[DIFF]], align 4, !dbg [[DBG7:![0-9]+]]
+; NONE-NEXT: br label %[[FOR_BODY:.*]], !dbg [[DBG8:![0-9]+]]
+; NONE: [[FOR_BODY]]:
+; NONE-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDVARS_IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; NONE-NEXT: [[ADD8:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[ADD:%.*]], %[[FOR_BODY]] ], !dbg [[DBG12:![0-9]+]]
+; NONE-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds [16 x i8], ptr [[CB]], i64 0, i64 [[INDVARS_IV]], !dbg [[DBG12]]
+; NONE-NEXT: [[TMP0:%.*]] = load i8, ptr [[ARRAYIDX]], align 1, !dbg [[DBG12]]
+; NONE-NEXT: [[CONV:%.*]] = sext i8 [[TMP0]] to i32, !dbg [[DBG12]]
+; NONE-NEXT: [[ARRAYIDX2:%.*]] = getelementptr inbounds [16 x i8], ptr [[CC]], i64 0, i64 [[INDVARS_IV]], !dbg [[DBG12]]
+; NONE-NEXT: [[TMP1:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1, !dbg [[DBG12]]
+; NONE-NEXT: [[CONV3:%.*]] = sext i8 [[TMP1]] to i32, !dbg [[DBG12]]
+; NONE-NEXT: [[SUB:%.*]] = sub i32 [[CONV]], [[CONV3]], !dbg [[DBG12]]
+; NONE-NEXT: [[ADD]] = add nsw i32 [[SUB]], [[ADD8]], !dbg [[DBG12]]
+; NONE-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1, !dbg [[DBG8]]
+; NONE-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], 16, !dbg [[DBG8]]
+; NONE-NEXT: br i1 [[EXITCOND]], label %[[FOR_END:.*]], label %[[FOR_BODY]], !dbg [[DBG8]]
+; NONE: [[FOR_END]]:
+; NONE-NEXT: [[ADD_LCSSA:%.*]] = phi i32 [ [[ADD]], %[[FOR_BODY]] ], !dbg [[DBG12]]
+; NONE-NEXT: store i32 [[ADD_LCSSA]], ptr [[DIFF]], align 4, !dbg [[DBG12]]
+; NONE-NEXT: call void @ibar(ptr [[DIFF]]), !dbg [[DBG14:![0-9]+]]
+; NONE-NEXT: ret i32 0, !dbg [[DBG15:![0-9]+]]
+;
entry:
%diff = alloca i32, align 4
%cb = alloca [16 x i8], align 16
@@ -60,3 +179,53 @@ declare void @ibar(ptr)
!22 = !DILocation(line: 20, column: 3, scope: !4)
!23 = !DILocation(line: 21, column: 3, scope: !4)
!24 = distinct !DICompileUnit(language: DW_LANG_C89, file: !1, emissionKind: NoDebug)
+;.
+; VECTORIZED: [[META2:![0-9]+]] = distinct !DICompileUnit(language: DW_LANG_C89, file: [[META3:![0-9]+]], isOptimized: false, runtimeVersion: 0, emissionKind: NoDebug)
+; VECTORIZED: [[META3]] = !DIFile(filename: "{{.*}}vectorization-remarks.c", directory: {{.*}})
+; VECTORIZED: [[DBG4]] = distinct !DISubprogram(name: "foo", scope: [[META3]], file: [[META3]], line: 5, type: [[META5:![0-9]+]], scopeLine: 6, virtualIndex: 6, flags: DIFlagPrototyped, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META2]], retainedNodes: [[META6:![0-9]+]])
+; VECTORIZED: [[META5]] = !DISubroutineType(types: [[META6]])
+; VECTORIZED: [[META6]] = !{}
+; VECTORIZED: [[DBG7]] = !DILocation(line: 8, column: 3, scope: [[DBG4]])
+; VECTORIZED: [[DBG8]] = !DILocation(line: 17, column: 8, scope: [[META9:![0-9]+]])
+; VECTORIZED: [[META9]] = distinct !DILexicalBlock(scope: [[META10:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; VECTORIZED: [[META10]] = distinct !DILexicalBlock(scope: [[META11:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; VECTORIZED: [[META11]] = distinct !DILexicalBlock(scope: [[DBG4]], file: [[META3]], line: 17, column: 3)
+; VECTORIZED: [[DBG12]] = !DILocation(line: 18, column: 5, scope: [[META13:![0-9]+]])
+; VECTORIZED: [[META13]] = distinct !DILexicalBlock(scope: [[META11]], file: [[META3]], line: 17, column: 27)
+; VECTORIZED: [[DBG14]] = !DILocation(line: 20, column: 3, scope: [[DBG4]])
+; VECTORIZED: [[DBG15]] = !DILocation(line: 21, column: 3, scope: [[DBG4]])
+;.
+; UNROLLED: [[META2:![0-9]+]] = distinct !DICompileUnit(language: DW_LANG_C89, file: [[META3:![0-9]+]], isOptimized: false, runtimeVersion: 0, emissionKind: NoDebug)
+; UNROLLED: [[META3]] = !DIFile(filename: "{{.*}}vectorization-remarks.c", directory: {{.*}})
+; UNROLLED: [[DBG4]] = distinct !DISubprogram(name: "foo", scope: [[META3]], file: [[META3]], line: 5, type: [[META5:![0-9]+]], scopeLine: 6, virtualIndex: 6, flags: DIFlagPrototyped, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META2]], retainedNodes: [[META6:![0-9]+]])
+; UNROLLED: [[META5]] = !DISubroutineType(types: [[META6]])
+; UNROLLED: [[META6]] = !{}
+; UNROLLED: [[DBG7]] = !DILocation(line: 8, column: 3, scope: [[DBG4]])
+; UNROLLED: [[DBG8]] = !DILocation(line: 17, column: 8, scope: [[META9:![0-9]+]])
+; UNROLLED: [[META9]] = distinct !DILexicalBlock(scope: [[META10:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; UNROLLED: [[META10]] = distinct !DILexicalBlock(scope: [[META11:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; UNROLLED: [[META11]] = distinct !DILexicalBlock(scope: [[DBG4]], file: [[META3]], line: 17, column: 3)
+; UNROLLED: [[DBG12]] = !DILocation(line: 18, column: 5, scope: [[META13:![0-9]+]])
+; UNROLLED: [[META13]] = distinct !DILexicalBlock(scope: [[META11]], file: [[META3]], line: 17, column: 27)
+; UNROLLED: [[LOOP14]] = distinct !{[[LOOP14]], [[META15:![0-9]+]], [[META16:![0-9]+]], [[META17:![0-9]+]]}
+; UNROLLED: [[META15]] = !{!"llvm.loop.isvectorized", i32 1}
+; UNROLLED: [[META16]] = !{!"llvm.loop.vectorize.body", i32 1}
+; UNROLLED: [[META17]] = !{!"llvm.loop.unroll.runtime.disable"}
+; UNROLLED: [[DBG18]] = !DILocation(line: 20, column: 3, scope: [[DBG4]])
+; UNROLLED: [[DBG19]] = !DILocation(line: 21, column: 3, scope: [[DBG4]])
+;.
+; NONE: [[META2:![0-9]+]] = distinct !DICompileUnit(language: DW_LANG_C89, file: [[META3:![0-9]+]], isOptimized: false, runtimeVersion: 0, emissionKind: NoDebug)
+; NONE: [[META3]] = !DIFile(filename: "{{.*}}vectorization-remarks.c", directory: {{.*}})
+; NONE: [[DBG4]] = distinct !DISubprogram(name: "foo", scope: [[META3]], file: [[META3]], line: 5, type: [[META5:![0-9]+]], scopeLine: 6, virtualIndex: 6, flags: DIFlagPrototyped, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META2]], retainedNodes: [[META6:![0-9]+]])
+; NONE: [[META5]] = !DISubroutineType(types: [[META6]])
+; NONE: [[META6]] = !{}
+; NONE: [[DBG7]] = !DILocation(line: 8, column: 3, scope: [[DBG4]])
+; NONE: [[DBG8]] = !DILocation(line: 17, column: 8, scope: [[META9:![0-9]+]])
+; NONE: [[META9]] = distinct !DILexicalBlock(scope: [[META10:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; NONE: [[META10]] = distinct !DILexicalBlock(scope: [[META11:![0-9]+]], file: [[META3]], line: 17, column: 8)
+; NONE: [[META11]] = distinct !DILexicalBlock(scope: [[DBG4]], file: [[META3]], line: 17, column: 3)
+; NONE: [[DBG12]] = !DILocation(line: 18, column: 5, scope: [[META13:![0-9]+]])
+; NONE: [[META13]] = distinct !DILexicalBlock(scope: [[META11]], file: [[META3]], line: 17, column: 27)
+; NONE: [[DBG14]] = !DILocation(line: 20, column: 3, scope: [[DBG4]])
+; NONE: [[DBG15]] = !DILocation(line: 21, column: 3, scope: [[DBG4]])
+;.
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/pixel-splat.ll b/llvm/test/Transforms/PhaseOrdering/X86/pixel-splat.ll
index 975de14a95fc4..d168f04485caf 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/pixel-splat.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/pixel-splat.ll
@@ -23,37 +23,56 @@ define void @loop_or(ptr noalias %pIn, ptr noalias %pOut, i32 %s) {
; CHECK-NEXT: entry:
; CHECK-NEXT: [[CMP1:%.*]] = icmp sgt i32 [[S:%.*]], 0
; CHECK-NEXT: br i1 [[CMP1]], label [[FOR_BODY_PREHEADER:%.*]], label [[FOR_END:%.*]]
-; CHECK: for.body.preheader:
+; CHECK: iter.check:
; CHECK-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext nneg i32 [[S]] to i64
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[S]], 8
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[S]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[FOR_BODY_PREHEADER5:%.*]], label [[VECTOR_PH:%.*]]
+; CHECK: vector.main.loop.iter.check:
+; CHECK-NEXT: [[MIN_ITERS_CHECK4:%.*]] = icmp ult i32 [[S]], 16
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK4]], label [[VEC_EPILOG_PH:%.*]], label [[VECTOR_PH1:%.*]]
; CHECK: vector.ph:
-; CHECK-NEXT: [[N_VEC:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 2147483640
+; CHECK-NEXT: [[TMP8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 12
+; CHECK-NEXT: [[N_VEC1:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 2147483632
; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
; CHECK: vector.body:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH1]] ], [ [[INDEX_NEXT1:%.*]], [[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[PIN:%.*]], i64 [[INDEX]]
-; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP0]], i64 4
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP2:%.*]] = zext <16 x i8> [[WIDE_LOAD]] to <16 x i32>
+; CHECK-NEXT: [[TMP12:%.*]] = mul nuw nsw <16 x i32> [[TMP2]], splat (i32 65793)
+; CHECK-NEXT: [[TMP4:%.*]] = or disjoint <16 x i32> [[TMP12]], splat (i32 -16777216)
+; CHECK-NEXT: [[TMP13:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[POUT:%.*]], i64 [[INDEX]]
+; CHECK-NEXT: store <16 x i32> [[TMP4]], ptr [[TMP13]], align 4
+; CHECK-NEXT: [[INDEX_NEXT1]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT1]], [[N_VEC1]]
+; CHECK-NEXT: br i1 [[TMP6]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: [[CMP_N1:%.*]] = icmp eq i64 [[N_VEC1]], [[WIDE_TRIP_COUNT]]
+; CHECK-NEXT: br i1 [[CMP_N1]], label [[FOR_END]], label [[VEC_EPILOG_ITER_CHECK:%.*]]
+; CHECK: vec.epilog.iter.check:
+; CHECK-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp eq i64 [[TMP8]], 0
+; CHECK-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label [[FOR_BODY_PREHEADER5]], label [[VEC_EPILOG_PH]], !prof [[PROF3:![0-9]+]]
+; CHECK: vec.epilog.ph:
+; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC1]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[VECTOR_PH]] ]
+; CHECK-NEXT: [[N_VEC:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 2147483644
+; CHECK-NEXT: br label [[VEC_EPILOG_VECTOR_BODY:%.*]]
+; CHECK: vec.epilog.vector.body:
+; CHECK-NEXT: [[INDEX6:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], [[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[PIN]], i64 [[INDEX6]]
; CHECK-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
-; CHECK-NEXT: [[TMP2:%.*]] = zext <4 x i8> [[WIDE_LOAD]] to <4 x i32>
; CHECK-NEXT: [[TMP3:%.*]] = zext <4 x i8> [[WIDE_LOAD4]] to <4 x i32>
-; CHECK-NEXT: [[TMP4:%.*]] = mul nuw nsw <4 x i32> [[TMP2]], splat (i32 65793)
; CHECK-NEXT: [[TMP5:%.*]] = mul nuw nsw <4 x i32> [[TMP3]], splat (i32 65793)
-; CHECK-NEXT: [[TMP6:%.*]] = or disjoint <4 x i32> [[TMP4]], splat (i32 -16777216)
; CHECK-NEXT: [[TMP7:%.*]] = or disjoint <4 x i32> [[TMP5]], splat (i32 -16777216)
-; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[POUT:%.*]], i64 [[INDEX]]
-; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP8]], i64 16
-; CHECK-NEXT: store <4 x i32> [[TMP6]], ptr [[TMP8]], align 4
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[POUT]], i64 [[INDEX6]]
; CHECK-NEXT: store <4 x i32> [[TMP7]], ptr [[TMP9]], align 4
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX6]], 4
; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT: br i1 [[TMP10]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK: middle.block:
+; CHECK-NEXT: br i1 [[TMP10]], label [[VEC_EPILOG_MIDDLE_BLOCK:%.*]], label [[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: vec.epilog.middle.block:
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N_VEC]], [[WIDE_TRIP_COUNT]]
; CHECK-NEXT: br i1 [[CMP_N]], label [[FOR_END]], label [[FOR_BODY_PREHEADER5]]
-; CHECK: for.body.preheader5:
-; CHECK-NEXT: [[INDVARS_IV_PH:%.*]] = phi i64 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[N_VEC]], [[MIDDLE_BLOCK]] ]
+; CHECK: for.body.preheader:
+; CHECK-NEXT: [[INDVARS_IV_PH:%.*]] = phi i64 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[N_VEC1]], [[VEC_EPILOG_ITER_CHECK]] ], [ [[N_VEC]], [[VEC_EPILOG_MIDDLE_BLOCK]] ]
; CHECK-NEXT: br label [[FOR_BODY:%.*]]
; CHECK: for.body:
; CHECK-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY]] ], [ [[INDVARS_IV_PH]], [[FOR_BODY_PREHEADER5]] ]
@@ -66,7 +85,7 @@ define void @loop_or(ptr noalias %pIn, ptr noalias %pOut, i32 %s) {
; CHECK-NEXT: store i32 [[OR3]], ptr [[ARRAYIDX5]], align 4
; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1
; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], [[WIDE_TRIP_COUNT]]
-; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label [[FOR_END]], label [[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label [[FOR_END]], label [[FOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: for.end:
; CHECK-NEXT: ret void
;
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/preserve-access-group.ll b/llvm/test/Transforms/PhaseOrdering/X86/preserve-access-group.ll
index a66528f1f12b4..9ca53b494de2d 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/preserve-access-group.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/preserve-access-group.ll
@@ -15,27 +15,27 @@ define void @test(i32 noundef %nface, i32 noundef %ncell, ptr noalias noundef %f
; CHECK: [[FOR_BODY_PREHEADER]]:
; CHECK-NEXT: [[TMP0:%.*]] = zext nneg i32 [[NFACE]] to i64
; CHECK-NEXT: [[INVARIANT_GEP:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[FACE_CELL]], i64 [[TMP0]]
-; CHECK-NEXT: [[TMP1:%.*]] = icmp ult i32 [[NFACE]], 4
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ult i32 [[NFACE]], 8
; CHECK-NEXT: br i1 [[TMP1]], label %[[FOR_BODY_PREHEADER14:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[UNROLL_ITER:%.*]] = and i64 [[TMP0]], 2147483644
+; CHECK-NEXT: [[UNROLL_ITER:%.*]] = and i64 [[TMP0]], 2147483640
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDVARS_IV_EPIL:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[FACE_CELL]], i64 [[INDVARS_IV_EPIL]]
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP10]], align 4, !tbaa [[INT_TBAA0:![0-9]+]], !llvm.access.group [[ACC_GRP4:![0-9]+]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP10]], align 4, !tbaa [[INT_TBAA0:![0-9]+]], !llvm.access.group [[ACC_GRP4:![0-9]+]]
; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[INVARIANT_GEP]], i64 [[INDVARS_IV_EPIL]]
-; CHECK-NEXT: [[WIDE_LOAD12:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !tbaa [[INT_TBAA0]], !llvm.access.group [[ACC_GRP4]]
-; CHECK-NEXT: [[TMP3:%.*]] = sext <4 x i32> [[WIDE_LOAD]] to <4 x i64>
-; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds [8 x i8], ptr [[Y]], <4 x i64> [[TMP3]]
-; CHECK-NEXT: [[TMP5:%.*]] = sext <4 x i32> [[WIDE_LOAD12]] to <4 x i64>
-; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds [8 x i8], ptr [[X]], <4 x i64> [[TMP5]]
-; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = tail call <4 x double> @llvm.masked.gather.v4f64.v4p0(<4 x ptr> align 8 [[TMP4]], <4 x i1> splat (i1 true), <4 x double> poison), !tbaa [[DOUBLE_TBAA5:![0-9]+]], !llvm.access.group [[ACC_GRP4]]
-; CHECK-NEXT: [[WIDE_MASKED_GATHER13:%.*]] = tail call <4 x double> @llvm.masked.gather.v4f64.v4p0(<4 x ptr> align 8 [[TMP6]], <4 x i1> splat (i1 true), <4 x double> poison), !tbaa [[DOUBLE_TBAA5]], !llvm.access.group [[ACC_GRP4]]
-; CHECK-NEXT: [[TMP7:%.*]] = fcmp fast olt <4 x double> [[WIDE_MASKED_GATHER]], [[WIDE_MASKED_GATHER13]]
-; CHECK-NEXT: [[TMP8:%.*]] = select <4 x i1> [[TMP7]], <4 x double> [[WIDE_MASKED_GATHER13]], <4 x double> [[WIDE_MASKED_GATHER]]
-; CHECK-NEXT: tail call void @llvm.masked.scatter.v4f64.v4p0(<4 x double> [[TMP8]], <4 x ptr> align 8 [[TMP4]], <4 x i1> splat (i1 true)), !tbaa [[DOUBLE_TBAA5]], !llvm.access.group [[ACC_GRP4]]
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDVARS_IV_EPIL]], 4
+; CHECK-NEXT: [[WIDE_LOAD12:%.*]] = load <8 x i32>, ptr [[TMP2]], align 4, !tbaa [[INT_TBAA0]], !llvm.access.group [[ACC_GRP4]]
+; CHECK-NEXT: [[TMP3:%.*]] = sext <8 x i32> [[WIDE_LOAD]] to <8 x i64>
+; CHECK-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds [8 x i8], ptr [[Y]], <8 x i64> [[TMP3]]
+; CHECK-NEXT: [[TMP4:%.*]] = sext <8 x i32> [[WIDE_LOAD12]] to <8 x i64>
+; CHECK-NEXT: [[WIDE_GEP13:%.*]] = getelementptr inbounds [8 x i8], ptr [[X]], <8 x i64> [[TMP4]]
+; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = tail call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 8 [[WIDE_GEP]], <8 x i1> splat (i1 true), <8 x double> poison), !tbaa [[DOUBLE_TBAA5:![0-9]+]], !llvm.access.group [[ACC_GRP4]]
+; CHECK-NEXT: [[WIDE_MASKED_GATHER14:%.*]] = tail call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 8 [[WIDE_GEP13]], <8 x i1> splat (i1 true), <8 x double> poison), !tbaa [[DOUBLE_TBAA5]], !llvm.access.group [[ACC_GRP4]]
+; CHECK-NEXT: [[TMP5:%.*]] = fcmp fast olt <8 x double> [[WIDE_MASKED_GATHER]], [[WIDE_MASKED_GATHER14]]
+; CHECK-NEXT: [[TMP6:%.*]] = select <8 x i1> [[TMP5]], <8 x double> [[WIDE_MASKED_GATHER14]], <8 x double> [[WIDE_MASKED_GATHER]]
+; CHECK-NEXT: tail call void @llvm.masked.scatter.v8f64.v8p0(<8 x double> [[TMP6]], <8 x ptr> align 8 [[WIDE_GEP]], <8 x i1> splat (i1 true)), !tbaa [[DOUBLE_TBAA5]], !llvm.access.group [[ACC_GRP4]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDVARS_IV_EPIL]], 8
; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[UNROLL_ITER]]
; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-known-first-value.ll b/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-known-first-value.ll
index 149dac30062cf..5de11ef4808ff 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-known-first-value.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/vector-reduction-known-first-value.ll
@@ -11,25 +11,261 @@ define i16 @test(ptr %ptr) {
; CHECK-NEXT: entry:
; CHECK-NEXT: [[FIRST:%.*]] = load i8, ptr [[PTR:%.*]], align 1
; CHECK-NEXT: tail call void @use(i8 [[FIRST]]) #[[ATTR2:[0-9]+]]
-; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
-; CHECK: vector.body:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <8 x i16> [ zeroinitializer, [[ENTRY]] ], [ [[TMP4:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI1:%.*]] = phi <8 x i16> [ zeroinitializer, [[ENTRY]] ], [ [[TMP5:%.*]], [[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 [[INDEX]]
-; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP0]], i64 8
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i8>, ptr [[TMP0]], align 1
-; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <8 x i8>, ptr [[TMP1]], align 1
-; CHECK-NEXT: [[TMP2:%.*]] = zext <8 x i8> [[WIDE_LOAD]] to <8 x i16>
-; CHECK-NEXT: [[TMP3:%.*]] = zext <8 x i8> [[WIDE_LOAD2]] to <8 x i16>
-; CHECK-NEXT: [[TMP4]] = add <8 x i16> [[VEC_PHI]], [[TMP2]]
-; CHECK-NEXT: [[TMP5]] = add <8 x i16> [[VEC_PHI1]], [[TMP3]]
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
-; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
-; CHECK-NEXT: br i1 [[TMP6]], label [[EXIT:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK: exit:
-; CHECK-NEXT: [[BIN_RDX:%.*]] = add <8 x i16> [[TMP5]], [[TMP4]]
-; CHECK-NEXT: [[TMP7:%.*]] = tail call i16 @llvm.vector.reduce.add.v8i16(<8 x i16> [[BIN_RDX]])
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 16
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[PTR]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <16 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = zext <16 x i8> [[WIDE_LOAD]] to <16 x i16>
+; CHECK-NEXT: [[TMP2:%.*]] = zext <16 x i8> [[WIDE_LOAD2]] to <16 x i16>
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 32
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 48
+; CHECK-NEXT: [[WIDE_LOAD_1:%.*]] = load <16 x i8>, ptr [[TMP3]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_1:%.*]] = load <16 x i8>, ptr [[TMP4]], align 1
+; CHECK-NEXT: [[TMP5:%.*]] = zext <16 x i8> [[WIDE_LOAD_1]] to <16 x i16>
+; CHECK-NEXT: [[TMP6:%.*]] = zext <16 x i8> [[WIDE_LOAD2_1]] to <16 x i16>
+; CHECK-NEXT: [[TMP189:%.*]] = add nuw nsw <16 x i16> [[TMP1]], [[TMP5]]
+; CHECK-NEXT: [[TMP8:%.*]] = add nuw nsw <16 x i16> [[TMP2]], [[TMP6]]
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 64
+; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 80
+; CHECK-NEXT: [[WIDE_LOAD_2:%.*]] = load <16 x i8>, ptr [[TMP9]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_2:%.*]] = load <16 x i8>, ptr [[TMP10]], align 1
+; CHECK-NEXT: [[TMP11:%.*]] = zext <16 x i8> [[WIDE_LOAD_2]] to <16 x i16>
+; CHECK-NEXT: [[TMP12:%.*]] = zext <16 x i8> [[WIDE_LOAD2_2]] to <16 x i16>
+; CHECK-NEXT: [[TMP13:%.*]] = add nuw nsw <16 x i16> [[TMP189]], [[TMP11]]
+; CHECK-NEXT: [[TMP14:%.*]] = add nuw nsw <16 x i16> [[TMP8]], [[TMP12]]
+; CHECK-NEXT: [[TMP15:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 96
+; CHECK-NEXT: [[TMP16:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 112
+; CHECK-NEXT: [[WIDE_LOAD_3:%.*]] = load <16 x i8>, ptr [[TMP15]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_3:%.*]] = load <16 x i8>, ptr [[TMP16]], align 1
+; CHECK-NEXT: [[TMP17:%.*]] = zext <16 x i8> [[WIDE_LOAD_3]] to <16 x i16>
+; CHECK-NEXT: [[TMP18:%.*]] = zext <16 x i8> [[WIDE_LOAD2_3]] to <16 x i16>
+; CHECK-NEXT: [[TMP19:%.*]] = add nuw nsw <16 x i16> [[TMP13]], [[TMP17]]
+; CHECK-NEXT: [[TMP20:%.*]] = add nuw nsw <16 x i16> [[TMP14]], [[TMP18]]
+; CHECK-NEXT: [[TMP21:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 128
+; CHECK-NEXT: [[TMP22:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 144
+; CHECK-NEXT: [[WIDE_LOAD_4:%.*]] = load <16 x i8>, ptr [[TMP21]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_4:%.*]] = load <16 x i8>, ptr [[TMP22]], align 1
+; CHECK-NEXT: [[TMP23:%.*]] = zext <16 x i8> [[WIDE_LOAD_4]] to <16 x i16>
+; CHECK-NEXT: [[TMP24:%.*]] = zext <16 x i8> [[WIDE_LOAD2_4]] to <16 x i16>
+; CHECK-NEXT: [[TMP25:%.*]] = add nuw nsw <16 x i16> [[TMP19]], [[TMP23]]
+; CHECK-NEXT: [[TMP26:%.*]] = add nuw nsw <16 x i16> [[TMP20]], [[TMP24]]
+; CHECK-NEXT: [[TMP27:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 160
+; CHECK-NEXT: [[TMP28:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 176
+; CHECK-NEXT: [[WIDE_LOAD_5:%.*]] = load <16 x i8>, ptr [[TMP27]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_5:%.*]] = load <16 x i8>, ptr [[TMP28]], align 1
+; CHECK-NEXT: [[TMP29:%.*]] = zext <16 x i8> [[WIDE_LOAD_5]] to <16 x i16>
+; CHECK-NEXT: [[TMP30:%.*]] = zext <16 x i8> [[WIDE_LOAD2_5]] to <16 x i16>
+; CHECK-NEXT: [[TMP31:%.*]] = add nuw nsw <16 x i16> [[TMP25]], [[TMP29]]
+; CHECK-NEXT: [[TMP32:%.*]] = add nuw nsw <16 x i16> [[TMP26]], [[TMP30]]
+; CHECK-NEXT: [[TMP33:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 192
+; CHECK-NEXT: [[TMP34:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 208
+; CHECK-NEXT: [[WIDE_LOAD_6:%.*]] = load <16 x i8>, ptr [[TMP33]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_6:%.*]] = load <16 x i8>, ptr [[TMP34]], align 1
+; CHECK-NEXT: [[TMP35:%.*]] = zext <16 x i8> [[WIDE_LOAD_6]] to <16 x i16>
+; CHECK-NEXT: [[TMP36:%.*]] = zext <16 x i8> [[WIDE_LOAD2_6]] to <16 x i16>
+; CHECK-NEXT: [[TMP37:%.*]] = add nuw nsw <16 x i16> [[TMP31]], [[TMP35]]
+; CHECK-NEXT: [[TMP38:%.*]] = add nuw nsw <16 x i16> [[TMP32]], [[TMP36]]
+; CHECK-NEXT: [[TMP39:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 224
+; CHECK-NEXT: [[TMP40:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 240
+; CHECK-NEXT: [[WIDE_LOAD_7:%.*]] = load <16 x i8>, ptr [[TMP39]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_7:%.*]] = load <16 x i8>, ptr [[TMP40]], align 1
+; CHECK-NEXT: [[TMP41:%.*]] = zext <16 x i8> [[WIDE_LOAD_7]] to <16 x i16>
+; CHECK-NEXT: [[TMP42:%.*]] = zext <16 x i8> [[WIDE_LOAD2_7]] to <16 x i16>
+; CHECK-NEXT: [[TMP43:%.*]] = add <16 x i16> [[TMP37]], [[TMP41]]
+; CHECK-NEXT: [[TMP44:%.*]] = add <16 x i16> [[TMP38]], [[TMP42]]
+; CHECK-NEXT: [[TMP45:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 256
+; CHECK-NEXT: [[TMP46:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 272
+; CHECK-NEXT: [[WIDE_LOAD_8:%.*]] = load <16 x i8>, ptr [[TMP45]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_8:%.*]] = load <16 x i8>, ptr [[TMP46]], align 1
+; CHECK-NEXT: [[TMP47:%.*]] = zext <16 x i8> [[WIDE_LOAD_8]] to <16 x i16>
+; CHECK-NEXT: [[TMP48:%.*]] = zext <16 x i8> [[WIDE_LOAD2_8]] to <16 x i16>
+; CHECK-NEXT: [[TMP49:%.*]] = add <16 x i16> [[TMP43]], [[TMP47]]
+; CHECK-NEXT: [[TMP50:%.*]] = add <16 x i16> [[TMP44]], [[TMP48]]
+; CHECK-NEXT: [[TMP51:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 288
+; CHECK-NEXT: [[TMP52:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 304
+; CHECK-NEXT: [[WIDE_LOAD_9:%.*]] = load <16 x i8>, ptr [[TMP51]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_9:%.*]] = load <16 x i8>, ptr [[TMP52]], align 1
+; CHECK-NEXT: [[TMP53:%.*]] = zext <16 x i8> [[WIDE_LOAD_9]] to <16 x i16>
+; CHECK-NEXT: [[TMP54:%.*]] = zext <16 x i8> [[WIDE_LOAD2_9]] to <16 x i16>
+; CHECK-NEXT: [[TMP55:%.*]] = add <16 x i16> [[TMP49]], [[TMP53]]
+; CHECK-NEXT: [[TMP56:%.*]] = add <16 x i16> [[TMP50]], [[TMP54]]
+; CHECK-NEXT: [[TMP57:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 320
+; CHECK-NEXT: [[TMP58:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 336
+; CHECK-NEXT: [[WIDE_LOAD_10:%.*]] = load <16 x i8>, ptr [[TMP57]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_10:%.*]] = load <16 x i8>, ptr [[TMP58]], align 1
+; CHECK-NEXT: [[TMP59:%.*]] = zext <16 x i8> [[WIDE_LOAD_10]] to <16 x i16>
+; CHECK-NEXT: [[TMP60:%.*]] = zext <16 x i8> [[WIDE_LOAD2_10]] to <16 x i16>
+; CHECK-NEXT: [[TMP61:%.*]] = add <16 x i16> [[TMP55]], [[TMP59]]
+; CHECK-NEXT: [[TMP62:%.*]] = add <16 x i16> [[TMP56]], [[TMP60]]
+; CHECK-NEXT: [[TMP63:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 352
+; CHECK-NEXT: [[TMP64:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 368
+; CHECK-NEXT: [[WIDE_LOAD_11:%.*]] = load <16 x i8>, ptr [[TMP63]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_11:%.*]] = load <16 x i8>, ptr [[TMP64]], align 1
+; CHECK-NEXT: [[TMP65:%.*]] = zext <16 x i8> [[WIDE_LOAD_11]] to <16 x i16>
+; CHECK-NEXT: [[TMP66:%.*]] = zext <16 x i8> [[WIDE_LOAD2_11]] to <16 x i16>
+; CHECK-NEXT: [[TMP67:%.*]] = add <16 x i16> [[TMP61]], [[TMP65]]
+; CHECK-NEXT: [[TMP68:%.*]] = add <16 x i16> [[TMP62]], [[TMP66]]
+; CHECK-NEXT: [[TMP69:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 384
+; CHECK-NEXT: [[TMP70:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 400
+; CHECK-NEXT: [[WIDE_LOAD_12:%.*]] = load <16 x i8>, ptr [[TMP69]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_12:%.*]] = load <16 x i8>, ptr [[TMP70]], align 1
+; CHECK-NEXT: [[TMP71:%.*]] = zext <16 x i8> [[WIDE_LOAD_12]] to <16 x i16>
+; CHECK-NEXT: [[TMP72:%.*]] = zext <16 x i8> [[WIDE_LOAD2_12]] to <16 x i16>
+; CHECK-NEXT: [[TMP73:%.*]] = add <16 x i16> [[TMP67]], [[TMP71]]
+; CHECK-NEXT: [[TMP74:%.*]] = add <16 x i16> [[TMP68]], [[TMP72]]
+; CHECK-NEXT: [[TMP75:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 416
+; CHECK-NEXT: [[TMP76:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 432
+; CHECK-NEXT: [[WIDE_LOAD_13:%.*]] = load <16 x i8>, ptr [[TMP75]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_13:%.*]] = load <16 x i8>, ptr [[TMP76]], align 1
+; CHECK-NEXT: [[TMP77:%.*]] = zext <16 x i8> [[WIDE_LOAD_13]] to <16 x i16>
+; CHECK-NEXT: [[TMP78:%.*]] = zext <16 x i8> [[WIDE_LOAD2_13]] to <16 x i16>
+; CHECK-NEXT: [[TMP79:%.*]] = add <16 x i16> [[TMP73]], [[TMP77]]
+; CHECK-NEXT: [[TMP80:%.*]] = add <16 x i16> [[TMP74]], [[TMP78]]
+; CHECK-NEXT: [[TMP81:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 448
+; CHECK-NEXT: [[TMP82:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 464
+; CHECK-NEXT: [[WIDE_LOAD_14:%.*]] = load <16 x i8>, ptr [[TMP81]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_14:%.*]] = load <16 x i8>, ptr [[TMP82]], align 1
+; CHECK-NEXT: [[TMP83:%.*]] = zext <16 x i8> [[WIDE_LOAD_14]] to <16 x i16>
+; CHECK-NEXT: [[TMP84:%.*]] = zext <16 x i8> [[WIDE_LOAD2_14]] to <16 x i16>
+; CHECK-NEXT: [[TMP85:%.*]] = add <16 x i16> [[TMP79]], [[TMP83]]
+; CHECK-NEXT: [[TMP86:%.*]] = add <16 x i16> [[TMP80]], [[TMP84]]
+; CHECK-NEXT: [[TMP87:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 480
+; CHECK-NEXT: [[TMP88:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 496
+; CHECK-NEXT: [[WIDE_LOAD_15:%.*]] = load <16 x i8>, ptr [[TMP87]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_15:%.*]] = load <16 x i8>, ptr [[TMP88]], align 1
+; CHECK-NEXT: [[TMP89:%.*]] = zext <16 x i8> [[WIDE_LOAD_15]] to <16 x i16>
+; CHECK-NEXT: [[TMP90:%.*]] = zext <16 x i8> [[WIDE_LOAD2_15]] to <16 x i16>
+; CHECK-NEXT: [[TMP91:%.*]] = add <16 x i16> [[TMP85]], [[TMP89]]
+; CHECK-NEXT: [[TMP92:%.*]] = add <16 x i16> [[TMP86]], [[TMP90]]
+; CHECK-NEXT: [[TMP93:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 512
+; CHECK-NEXT: [[TMP94:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 528
+; CHECK-NEXT: [[WIDE_LOAD_16:%.*]] = load <16 x i8>, ptr [[TMP93]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_16:%.*]] = load <16 x i8>, ptr [[TMP94]], align 1
+; CHECK-NEXT: [[TMP95:%.*]] = zext <16 x i8> [[WIDE_LOAD_16]] to <16 x i16>
+; CHECK-NEXT: [[TMP96:%.*]] = zext <16 x i8> [[WIDE_LOAD2_16]] to <16 x i16>
+; CHECK-NEXT: [[TMP97:%.*]] = add <16 x i16> [[TMP91]], [[TMP95]]
+; CHECK-NEXT: [[TMP98:%.*]] = add <16 x i16> [[TMP92]], [[TMP96]]
+; CHECK-NEXT: [[TMP99:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 544
+; CHECK-NEXT: [[TMP100:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 560
+; CHECK-NEXT: [[WIDE_LOAD_17:%.*]] = load <16 x i8>, ptr [[TMP99]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_17:%.*]] = load <16 x i8>, ptr [[TMP100]], align 1
+; CHECK-NEXT: [[TMP101:%.*]] = zext <16 x i8> [[WIDE_LOAD_17]] to <16 x i16>
+; CHECK-NEXT: [[TMP102:%.*]] = zext <16 x i8> [[WIDE_LOAD2_17]] to <16 x i16>
+; CHECK-NEXT: [[TMP103:%.*]] = add <16 x i16> [[TMP97]], [[TMP101]]
+; CHECK-NEXT: [[TMP104:%.*]] = add <16 x i16> [[TMP98]], [[TMP102]]
+; CHECK-NEXT: [[TMP105:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 576
+; CHECK-NEXT: [[TMP106:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 592
+; CHECK-NEXT: [[WIDE_LOAD_18:%.*]] = load <16 x i8>, ptr [[TMP105]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_18:%.*]] = load <16 x i8>, ptr [[TMP106]], align 1
+; CHECK-NEXT: [[TMP107:%.*]] = zext <16 x i8> [[WIDE_LOAD_18]] to <16 x i16>
+; CHECK-NEXT: [[TMP108:%.*]] = zext <16 x i8> [[WIDE_LOAD2_18]] to <16 x i16>
+; CHECK-NEXT: [[TMP109:%.*]] = add <16 x i16> [[TMP103]], [[TMP107]]
+; CHECK-NEXT: [[TMP110:%.*]] = add <16 x i16> [[TMP104]], [[TMP108]]
+; CHECK-NEXT: [[TMP111:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 608
+; CHECK-NEXT: [[TMP112:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 624
+; CHECK-NEXT: [[WIDE_LOAD_19:%.*]] = load <16 x i8>, ptr [[TMP111]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_19:%.*]] = load <16 x i8>, ptr [[TMP112]], align 1
+; CHECK-NEXT: [[TMP113:%.*]] = zext <16 x i8> [[WIDE_LOAD_19]] to <16 x i16>
+; CHECK-NEXT: [[TMP114:%.*]] = zext <16 x i8> [[WIDE_LOAD2_19]] to <16 x i16>
+; CHECK-NEXT: [[TMP115:%.*]] = add <16 x i16> [[TMP109]], [[TMP113]]
+; CHECK-NEXT: [[TMP116:%.*]] = add <16 x i16> [[TMP110]], [[TMP114]]
+; CHECK-NEXT: [[TMP117:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 640
+; CHECK-NEXT: [[TMP118:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 656
+; CHECK-NEXT: [[WIDE_LOAD_20:%.*]] = load <16 x i8>, ptr [[TMP117]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_20:%.*]] = load <16 x i8>, ptr [[TMP118]], align 1
+; CHECK-NEXT: [[TMP119:%.*]] = zext <16 x i8> [[WIDE_LOAD_20]] to <16 x i16>
+; CHECK-NEXT: [[TMP120:%.*]] = zext <16 x i8> [[WIDE_LOAD2_20]] to <16 x i16>
+; CHECK-NEXT: [[TMP121:%.*]] = add <16 x i16> [[TMP115]], [[TMP119]]
+; CHECK-NEXT: [[TMP122:%.*]] = add <16 x i16> [[TMP116]], [[TMP120]]
+; CHECK-NEXT: [[TMP123:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 672
+; CHECK-NEXT: [[TMP124:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 688
+; CHECK-NEXT: [[WIDE_LOAD_21:%.*]] = load <16 x i8>, ptr [[TMP123]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_21:%.*]] = load <16 x i8>, ptr [[TMP124]], align 1
+; CHECK-NEXT: [[TMP125:%.*]] = zext <16 x i8> [[WIDE_LOAD_21]] to <16 x i16>
+; CHECK-NEXT: [[TMP126:%.*]] = zext <16 x i8> [[WIDE_LOAD2_21]] to <16 x i16>
+; CHECK-NEXT: [[TMP127:%.*]] = add <16 x i16> [[TMP121]], [[TMP125]]
+; CHECK-NEXT: [[TMP128:%.*]] = add <16 x i16> [[TMP122]], [[TMP126]]
+; CHECK-NEXT: [[TMP129:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 704
+; CHECK-NEXT: [[TMP130:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 720
+; CHECK-NEXT: [[WIDE_LOAD_22:%.*]] = load <16 x i8>, ptr [[TMP129]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_22:%.*]] = load <16 x i8>, ptr [[TMP130]], align 1
+; CHECK-NEXT: [[TMP131:%.*]] = zext <16 x i8> [[WIDE_LOAD_22]] to <16 x i16>
+; CHECK-NEXT: [[TMP132:%.*]] = zext <16 x i8> [[WIDE_LOAD2_22]] to <16 x i16>
+; CHECK-NEXT: [[TMP133:%.*]] = add <16 x i16> [[TMP127]], [[TMP131]]
+; CHECK-NEXT: [[TMP134:%.*]] = add <16 x i16> [[TMP128]], [[TMP132]]
+; CHECK-NEXT: [[TMP135:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 736
+; CHECK-NEXT: [[TMP136:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 752
+; CHECK-NEXT: [[WIDE_LOAD_23:%.*]] = load <16 x i8>, ptr [[TMP135]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_23:%.*]] = load <16 x i8>, ptr [[TMP136]], align 1
+; CHECK-NEXT: [[TMP137:%.*]] = zext <16 x i8> [[WIDE_LOAD_23]] to <16 x i16>
+; CHECK-NEXT: [[TMP138:%.*]] = zext <16 x i8> [[WIDE_LOAD2_23]] to <16 x i16>
+; CHECK-NEXT: [[TMP139:%.*]] = add <16 x i16> [[TMP133]], [[TMP137]]
+; CHECK-NEXT: [[TMP140:%.*]] = add <16 x i16> [[TMP134]], [[TMP138]]
+; CHECK-NEXT: [[TMP141:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 768
+; CHECK-NEXT: [[TMP142:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 784
+; CHECK-NEXT: [[WIDE_LOAD_24:%.*]] = load <16 x i8>, ptr [[TMP141]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_24:%.*]] = load <16 x i8>, ptr [[TMP142]], align 1
+; CHECK-NEXT: [[TMP143:%.*]] = zext <16 x i8> [[WIDE_LOAD_24]] to <16 x i16>
+; CHECK-NEXT: [[TMP144:%.*]] = zext <16 x i8> [[WIDE_LOAD2_24]] to <16 x i16>
+; CHECK-NEXT: [[TMP145:%.*]] = add <16 x i16> [[TMP139]], [[TMP143]]
+; CHECK-NEXT: [[TMP146:%.*]] = add <16 x i16> [[TMP140]], [[TMP144]]
+; CHECK-NEXT: [[TMP147:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 800
+; CHECK-NEXT: [[TMP148:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 816
+; CHECK-NEXT: [[WIDE_LOAD_25:%.*]] = load <16 x i8>, ptr [[TMP147]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_25:%.*]] = load <16 x i8>, ptr [[TMP148]], align 1
+; CHECK-NEXT: [[TMP149:%.*]] = zext <16 x i8> [[WIDE_LOAD_25]] to <16 x i16>
+; CHECK-NEXT: [[TMP150:%.*]] = zext <16 x i8> [[WIDE_LOAD2_25]] to <16 x i16>
+; CHECK-NEXT: [[TMP151:%.*]] = add <16 x i16> [[TMP145]], [[TMP149]]
+; CHECK-NEXT: [[TMP152:%.*]] = add <16 x i16> [[TMP146]], [[TMP150]]
+; CHECK-NEXT: [[TMP153:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 832
+; CHECK-NEXT: [[TMP154:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 848
+; CHECK-NEXT: [[WIDE_LOAD_26:%.*]] = load <16 x i8>, ptr [[TMP153]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_26:%.*]] = load <16 x i8>, ptr [[TMP154]], align 1
+; CHECK-NEXT: [[TMP155:%.*]] = zext <16 x i8> [[WIDE_LOAD_26]] to <16 x i16>
+; CHECK-NEXT: [[TMP156:%.*]] = zext <16 x i8> [[WIDE_LOAD2_26]] to <16 x i16>
+; CHECK-NEXT: [[TMP157:%.*]] = add <16 x i16> [[TMP151]], [[TMP155]]
+; CHECK-NEXT: [[TMP158:%.*]] = add <16 x i16> [[TMP152]], [[TMP156]]
+; CHECK-NEXT: [[TMP159:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 864
+; CHECK-NEXT: [[TMP160:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 880
+; CHECK-NEXT: [[WIDE_LOAD_27:%.*]] = load <16 x i8>, ptr [[TMP159]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_27:%.*]] = load <16 x i8>, ptr [[TMP160]], align 1
+; CHECK-NEXT: [[TMP161:%.*]] = zext <16 x i8> [[WIDE_LOAD_27]] to <16 x i16>
+; CHECK-NEXT: [[TMP162:%.*]] = zext <16 x i8> [[WIDE_LOAD2_27]] to <16 x i16>
+; CHECK-NEXT: [[TMP163:%.*]] = add <16 x i16> [[TMP157]], [[TMP161]]
+; CHECK-NEXT: [[TMP164:%.*]] = add <16 x i16> [[TMP158]], [[TMP162]]
+; CHECK-NEXT: [[TMP165:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 896
+; CHECK-NEXT: [[TMP166:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 912
+; CHECK-NEXT: [[WIDE_LOAD_28:%.*]] = load <16 x i8>, ptr [[TMP165]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_28:%.*]] = load <16 x i8>, ptr [[TMP166]], align 1
+; CHECK-NEXT: [[TMP167:%.*]] = zext <16 x i8> [[WIDE_LOAD_28]] to <16 x i16>
+; CHECK-NEXT: [[TMP168:%.*]] = zext <16 x i8> [[WIDE_LOAD2_28]] to <16 x i16>
+; CHECK-NEXT: [[TMP169:%.*]] = add <16 x i16> [[TMP163]], [[TMP167]]
+; CHECK-NEXT: [[TMP170:%.*]] = add <16 x i16> [[TMP164]], [[TMP168]]
+; CHECK-NEXT: [[TMP171:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 928
+; CHECK-NEXT: [[TMP172:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 944
+; CHECK-NEXT: [[WIDE_LOAD_29:%.*]] = load <16 x i8>, ptr [[TMP171]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_29:%.*]] = load <16 x i8>, ptr [[TMP172]], align 1
+; CHECK-NEXT: [[TMP173:%.*]] = zext <16 x i8> [[WIDE_LOAD_29]] to <16 x i16>
+; CHECK-NEXT: [[TMP174:%.*]] = zext <16 x i8> [[WIDE_LOAD2_29]] to <16 x i16>
+; CHECK-NEXT: [[TMP175:%.*]] = add <16 x i16> [[TMP169]], [[TMP173]]
+; CHECK-NEXT: [[TMP176:%.*]] = add <16 x i16> [[TMP170]], [[TMP174]]
+; CHECK-NEXT: [[TMP177:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 960
+; CHECK-NEXT: [[TMP178:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 976
+; CHECK-NEXT: [[WIDE_LOAD_30:%.*]] = load <16 x i8>, ptr [[TMP177]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_30:%.*]] = load <16 x i8>, ptr [[TMP178]], align 1
+; CHECK-NEXT: [[TMP179:%.*]] = zext <16 x i8> [[WIDE_LOAD_30]] to <16 x i16>
+; CHECK-NEXT: [[TMP180:%.*]] = zext <16 x i8> [[WIDE_LOAD2_30]] to <16 x i16>
+; CHECK-NEXT: [[TMP181:%.*]] = add <16 x i16> [[TMP175]], [[TMP179]]
+; CHECK-NEXT: [[TMP182:%.*]] = add <16 x i16> [[TMP176]], [[TMP180]]
+; CHECK-NEXT: [[TMP183:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 992
+; CHECK-NEXT: [[TMP184:%.*]] = getelementptr inbounds nuw i8, ptr [[PTR]], i64 1008
+; CHECK-NEXT: [[WIDE_LOAD_31:%.*]] = load <16 x i8>, ptr [[TMP183]], align 1
+; CHECK-NEXT: [[WIDE_LOAD2_31:%.*]] = load <16 x i8>, ptr [[TMP184]], align 1
+; CHECK-NEXT: [[TMP185:%.*]] = zext <16 x i8> [[WIDE_LOAD_31]] to <16 x i16>
+; CHECK-NEXT: [[TMP186:%.*]] = zext <16 x i8> [[WIDE_LOAD2_31]] to <16 x i16>
+; CHECK-NEXT: [[TMP187:%.*]] = add <16 x i16> [[TMP181]], [[TMP185]]
+; CHECK-NEXT: [[TMP188:%.*]] = add <16 x i16> [[TMP182]], [[TMP186]]
+; CHECK-NEXT: [[BIN_RDX:%.*]] = add <16 x i16> [[TMP188]], [[TMP187]]
+; CHECK-NEXT: [[TMP7:%.*]] = tail call i16 @llvm.vector.reduce.add.v16i16(<16 x i16> [[BIN_RDX]])
; CHECK-NEXT: ret i16 [[TMP7]]
;
entry:
>From c9126804d27493c8b0c5a69196ccfa269425d8fa Mon Sep 17 00:00:00 2001
From: Sumukh Bharadwaj <Sumukh.Bharadwaj at amd.com>
Date: Mon, 7 Sep 2026 14:39:01 +0530
Subject: [PATCH 2/2] [X86][CostModel] Price predicate mask-expansion fanout
for wide predicated ops
A masked memory op, gather/scatter or select produces its predicate as a
compact <N x i1>, but when the value/data type legalizes into more
register parts than that predicate, codegen must materialize a sub-mask
for every extra data part -- a kshiftr out of the k-register on AVX-512,
or a vector unpack/sign-extend on SSE/AVX. The per-part cost tables price
these ops linearly in data parts and miss the fanout, so per-lane cost
stays flat as the type widens. With MaximizeBandwidth sizing VF from the
narrowest type and the vectorizer tie-breaking toward the widest VF,
nothing penalizes over-widening and predicated loops blow up -- e.g. a
conditional store picking VF32 on AVX2 (a long vmaskmov chain) or VF64 on
AVX-512 (a long kshiftr chain).
Charge an additive (DataParts - MaskParts) * 4 for the extra parts:
* getMaskedMemoryOpCost and getGatherScatterOpCost on all subtargets --
the charge only fires on genuinely predicated memory ops, so it is
safe to apply everywhere and it fixes the AVX2 over-widening reported
on the MaximizeBandwidth enablement, not just the AVX-512 case.
* getCmpSelInstrCost gated to AVX-512 only. A vector select is a
pervasive primitive; on SSE/AVX its blend is already priced by the
per-part tables and adding the fanout over-penalizes ordinary,
non-predicated-memory loops and needlessly shrinks their VF (observed
on baseline SSE2). On AVX-512 the k-register -> wide-data kshiftr is a
real cost the tables miss, so the charge is kept there.
The penalty is zero whenever data and mask occupy the same number of
register parts, so maskless and single-part reductions -- the
MaximizeBandwidth throughput wins -- are unaffected. With this the
motivating conditional-store loop selects VF16 on AVX-512 (no kshiftr)
and VF4 on AVX2 (no over-wide vmaskmov chain), so no generic
vectorizer-level cost floor is required.
---
.../lib/Target/X86/X86TargetTransformInfo.cpp | 92 ++-
llvm/test/Analysis/CostModel/X86/fptoi_sat.ll | 12 +-
.../X86/gather-scatter-mask-expansion.ll | 40 ++
.../X86/masked-intrinsic-cost-inseltpoison.ll | 112 +--
.../CostModel/X86/masked-intrinsic-cost.ll | 112 +--
.../X86/masked-mem-mask-expansion.ll | 89 +++
.../CostModel/X86/select-mask-expansion.ll | 63 ++
.../X86/CostModel/gather-i32-with-i8-index.ll | 8 +-
.../X86/CostModel/gather-i64-with-i8-index.ll | 12 +-
.../interleaved-load-f64-stride-5.ll | 30 +-
.../interleaved-load-f64-stride-6.ll | 36 +-
.../interleaved-load-f64-stride-7.ll | 42 +-
.../interleaved-load-f64-stride-8.ll | 48 +-
.../interleaved-load-i64-stride-2.ll | 8 +-
.../interleaved-load-i64-stride-4.ll | 24 +-
.../interleaved-load-i64-stride-5.ll | 30 +-
.../interleaved-load-i64-stride-6.ll | 36 +-
.../interleaved-load-i64-stride-7.ll | 42 +-
.../interleaved-load-i64-stride-8.ll | 48 +-
.../interleaved-store-f64-stride-8.ll | 48 +-
.../interleaved-store-i64-stride-8.ll | 48 +-
.../masked-gather-i32-with-i8-index.ll | 8 +-
.../masked-gather-i64-with-i8-index.ll | 12 +-
.../X86/CostModel/masked-load-i16.ll | 2 +-
.../X86/CostModel/masked-load-i32.ll | 12 +-
.../X86/CostModel/masked-load-i64.ll | 18 +-
.../masked-scatter-i32-with-i8-index.ll | 4 +-
.../masked-scatter-i64-with-i8-index.ll | 6 +-
.../X86/CostModel/masked-store-i16.ll | 2 +-
.../X86/CostModel/masked-store-i32.ll | 12 +-
.../X86/CostModel/masked-store-i64.ll | 18 +-
.../CostModel/scatter-i32-with-i8-index.ll | 4 +-
.../CostModel/scatter-i64-with-i8-index.ll | 6 +-
.../LoopVectorize/X86/masked_load_store.ll | 635 ++++++++++++------
34 files changed, 1089 insertions(+), 630 deletions(-)
create mode 100644 llvm/test/Analysis/CostModel/X86/gather-scatter-mask-expansion.ll
create mode 100644 llvm/test/Analysis/CostModel/X86/masked-mem-mask-expansion.ll
create mode 100644 llvm/test/Analysis/CostModel/X86/select-mask-expansion.ll
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
index 86f7ee89a68da..6919bf52b28f3 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
@@ -3492,6 +3492,40 @@ InstructionCost X86TTIImpl::getCastInstrCost(unsigned Opcode, Type *Dst,
BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I));
}
+// Additive cost of "predicate fanout" (mask expansion) for a predicated op
+// whose value/data operand legalizes into more register parts than its mask.
+//
+// A predicated op (select, masked load/store, gather/scatter) produces its
+// mask as a compact <N x i1>: a compare result kept in one k-register on
+// AVX-512, or a single narrow vector on SSE/AVX. When the data type legalizes
+// into more register parts than that mask, codegen has to materialize a
+// sub-mask for every extra data part -- a kshiftr out of the k-register on
+// AVX-512, or a vector unpack/sign-extend on SSE/AVX -- each feeding a separate
+// per-part op. The per-part cost tables price the op linearly in data parts
+// and miss this fanout, so per-lane cost stays flat as the type widens; with
+// MaximizeBandwidth sizing VF from the narrowest type, nothing then penalizes
+// over-widening and predicated loops blow up.
+//
+// Charge FanoutCostPerPart per extra part. Two caveats worth stating:
+// - The metric assumes the i1 mask legalizes into no more parts than the
+// data; that holds on X86 (masks come from compares, kept compact). A
+// MaskParts of 0 (unexpected legalization) disables the charge.
+// - The constant is not an absolute latency. It only has to be large enough
+// that the part-matched VF beats the loop vectorizer's tie-break toward the
+// widest VF (a value of 2 tied and lost). It is orthogonal to the
+// promotion / expansion mask shuffles priced in getMaskedMemoryOpCost,
+// which cover data promotion and padding the mask to the legalized element
+// count -- not per-part sub-mask distribution.
+static InstructionCost getPredicateFanoutCost(const X86TTIImpl &TTI,
+ Type *DataVTy, Type *MaskVTy) {
+ constexpr unsigned FanoutCostPerPart = 4;
+ unsigned DataParts = TTI.getNumberOfParts(DataVTy);
+ unsigned MaskParts = TTI.getNumberOfParts(MaskVTy);
+ if (DataParts > MaskParts && MaskParts > 0)
+ return InstructionCost((DataParts - MaskParts) * FanoutCostPerPart);
+ return InstructionCost(0);
+}
+
InstructionCost X86TTIImpl::getCmpSelInstrCost(
unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info,
@@ -3509,6 +3543,18 @@ InstructionCost X86TTIImpl::getCmpSelInstrCost(
int ISD = TLI->InstructionOpcodeToISD(Opcode);
assert(ISD && "Invalid opcode");
+ // Mask-expansion (predicate fanout); see getPredicateFanoutCost. Gated to
+ // AVX-512 on purpose: a vector select is pervasive and on SSE/AVX its blend
+ // mask is already priced by the per-part tables, so charging the fanout there
+ // over-penalizes ordinary (non predicated-memory) loops and shrinks their VF.
+ // On AVX-512 the k-register -> wide-data kshiftr is a real cost the tables
+ // miss. The masked memory / gather / scatter fanout is charged on all
+ // subtargets because it fires only on genuinely predicated memory ops.
+ InstructionCost MaskExpCost = 0;
+ if (Opcode == Instruction::Select && ST->hasAVX512() &&
+ isa<VectorType>(ValTy) && isa_and_nonnull<VectorType>(CondTy))
+ MaskExpCost = getPredicateFanoutCost(*this, ValTy, CondTy);
+
InstructionCost ExtraCost = 0;
if (Opcode == Instruction::ICmp || Opcode == Instruction::FCmp) {
// Some vector comparison predicates cost extra instructions.
@@ -3735,52 +3781,52 @@ InstructionCost X86TTIImpl::getCmpSelInstrCost(
if (ST->useSLMArithCosts())
if (const auto *Entry = CostTableLookup(SLMCostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasBWI())
if (const auto *Entry = CostTableLookup(AVX512BWCostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasAVX512())
if (const auto *Entry = CostTableLookup(AVX512CostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasAVX2())
if (const auto *Entry = CostTableLookup(AVX2CostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasXOP())
if (const auto *Entry = CostTableLookup(XOPCostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasAVX())
if (const auto *Entry = CostTableLookup(AVX1CostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasSSE42())
if (const auto *Entry = CostTableLookup(SSE42CostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasSSE41())
if (const auto *Entry = CostTableLookup(SSE41CostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasSSE2())
if (const auto *Entry = CostTableLookup(SSE2CostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
if (ST->hasSSE1())
if (const auto *Entry = CostTableLookup(SSE1CostTbl, ISD, MTy))
if (auto KindCost = Entry->Cost[CostKind])
- return LT.first * (ExtraCost + *KindCost);
+ return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
// Assume a 3cy latency for fp select ops.
if (CostKind == TTI::TCK_Latency && Opcode == Instruction::Select)
@@ -3788,7 +3834,8 @@ InstructionCost X86TTIImpl::getCmpSelInstrCost(
return 3;
return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
- Op1Info, Op2Info, I);
+ Op1Info, Op2Info, I) +
+ MaskExpCost;
}
unsigned X86TTIImpl::getAtomicMemIntrinsicMaxElementSize() const { return 16; }
@@ -5679,12 +5726,18 @@ X86TTIImpl::getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA,
CostKind, 0, MaskTy);
}
+ // Mask-expansion (predicate fanout); see getPredicateFanoutCost.
+ auto *MaskVecTy =
+ FixedVectorType::get(Type::getInt1Ty(SrcVTy->getContext()), NumElem);
+ InstructionCost MaskExpCost =
+ getPredicateFanoutCost(*this, SrcVTy, MaskVecTy);
+
// Pre-AVX512 - each maskmov load costs 2 + store costs ~8.
if (!ST->hasAVX512())
- return Cost + LT.first * (IsLoad ? 2 : 8);
+ return Cost + LT.first * (IsLoad ? 2 : 8) + MaskExpCost;
- // AVX-512 masked load/store is cheaper
- return Cost + LT.first;
+ // AVX-512 masked load/store is cheaper.
+ return Cost + LT.first + MaskExpCost;
}
InstructionCost X86TTIImpl::getPointersChainCost(
@@ -6701,8 +6754,15 @@ X86TTIImpl::getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA,
assert(SrcVTy->isVectorTy() && "Unexpected data type for Gather/Scatter");
unsigned AddressSpace = MICA.getAddressSpace();
- return getGSVectorCost(Opcode, CostKind, SrcVTy, Ptr, Alignment,
- AddressSpace);
+ InstructionCost Cost =
+ getGSVectorCost(Opcode, CostKind, SrcVTy, Ptr, Alignment, AddressSpace);
+
+ // Mask-expansion (predicate fanout); see getPredicateFanoutCost.
+ auto *MaskVecTy =
+ FixedVectorType::get(Type::getInt1Ty(SrcVTy->getContext()),
+ cast<FixedVectorType>(SrcVTy)->getNumElements());
+ Cost += getPredicateFanoutCost(*this, SrcVTy, MaskVecTy);
+ return Cost;
}
bool X86TTIImpl::isLSRCostLess(const TargetTransformInfo::LSRCost &C1,
diff --git a/llvm/test/Analysis/CostModel/X86/fptoi_sat.ll b/llvm/test/Analysis/CostModel/X86/fptoi_sat.ll
index 41bf88b1ec316..e5df2badfeb34 100644
--- a/llvm/test/Analysis/CostModel/X86/fptoi_sat.ll
+++ b/llvm/test/Analysis/CostModel/X86/fptoi_sat.ll
@@ -512,7 +512,7 @@ define void @casts() {
; AVX512F-NEXT: Cost Model: Found an estimated cost of 11 for instruction: %v16f32u16 = call <16 x i16> @llvm.fptoui.sat.v16i16.v16f32(<16 x float> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 11 for instruction: %v16f32s32 = call <16 x i32> @llvm.fptosi.sat.v16i32.v16f32(<16 x float> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %v16f32u32 = call <16 x i32> @llvm.fptoui.sat.v16i32.v16f32(<16 x float> undef)
-; AVX512F-NEXT: Cost Model: Found an estimated cost of 72 for instruction: %v16f32s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f32(<16 x float> undef)
+; AVX512F-NEXT: Cost Model: Found an estimated cost of 76 for instruction: %v16f32s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f32(<16 x float> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 69 for instruction: %v16f32u64 = call <16 x i64> @llvm.fptoui.sat.v16i64.v16f32(<16 x float> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 49 for instruction: %v16f64s1 = call <16 x i1> @llvm.fptosi.sat.v16i1.v16f64(<16 x double> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 15 for instruction: %v16f64u1 = call <16 x i1> @llvm.fptoui.sat.v16i1.v16f64(<16 x double> undef)
@@ -522,7 +522,7 @@ define void @casts() {
; AVX512F-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %v16f64u16 = call <16 x i16> @llvm.fptoui.sat.v16i16.v16f64(<16 x double> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 18 for instruction: %v16f64s32 = call <16 x i32> @llvm.fptosi.sat.v16i32.v16f64(<16 x double> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 15 for instruction: %v16f64u32 = call <16 x i32> @llvm.fptoui.sat.v16i32.v16f64(<16 x double> undef)
-; AVX512F-NEXT: Cost Model: Found an estimated cost of 76 for instruction: %v16f64s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f64(<16 x double> undef)
+; AVX512F-NEXT: Cost Model: Found an estimated cost of 80 for instruction: %v16f64s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f64(<16 x double> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 72 for instruction: %v16f64u64 = call <16 x i64> @llvm.fptoui.sat.v16i64.v16f64(<16 x double> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
;
@@ -615,7 +615,7 @@ define void @casts() {
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 11 for instruction: %v16f32u16 = call <16 x i16> @llvm.fptoui.sat.v16i16.v16f32(<16 x float> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 11 for instruction: %v16f32s32 = call <16 x i32> @llvm.fptosi.sat.v16i32.v16f32(<16 x float> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 9 for instruction: %v16f32u32 = call <16 x i32> @llvm.fptoui.sat.v16i32.v16f32(<16 x float> undef)
-; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 14 for instruction: %v16f32s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f32(<16 x float> undef)
+; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 18 for instruction: %v16f32s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f32(<16 x float> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 11 for instruction: %v16f32u64 = call <16 x i64> @llvm.fptoui.sat.v16i64.v16f32(<16 x float> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 49 for instruction: %v16f64s1 = call <16 x i1> @llvm.fptosi.sat.v16i1.v16f64(<16 x double> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 15 for instruction: %v16f64u1 = call <16 x i1> @llvm.fptoui.sat.v16i1.v16f64(<16 x double> undef)
@@ -625,7 +625,7 @@ define void @casts() {
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %v16f64u16 = call <16 x i16> @llvm.fptoui.sat.v16i16.v16f64(<16 x double> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 18 for instruction: %v16f64s32 = call <16 x i32> @llvm.fptosi.sat.v16i32.v16f64(<16 x double> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 15 for instruction: %v16f64u32 = call <16 x i32> @llvm.fptoui.sat.v16i32.v16f64(<16 x double> undef)
-; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 18 for instruction: %v16f64s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f64(<16 x double> undef)
+; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 22 for instruction: %v16f64s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f64(<16 x double> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 14 for instruction: %v16f64u64 = call <16 x i64> @llvm.fptoui.sat.v16i64.v16f64(<16 x double> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
;
@@ -1054,7 +1054,7 @@ define void @fp16() {
; AVX512F-NEXT: Cost Model: Found an estimated cost of 178 for instruction: %v16f16u16 = call <16 x i16> @llvm.fptoui.sat.v16i16.v16f16(<16 x half> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 178 for instruction: %v16f16s32 = call <16 x i32> @llvm.fptosi.sat.v16i32.v16f16(<16 x half> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 176 for instruction: %v16f16u32 = call <16 x i32> @llvm.fptoui.sat.v16i32.v16f16(<16 x half> undef)
-; AVX512F-NEXT: Cost Model: Found an estimated cost of 186 for instruction: %v16f16s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f16(<16 x half> undef)
+; AVX512F-NEXT: Cost Model: Found an estimated cost of 190 for instruction: %v16f16s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f16(<16 x half> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 183 for instruction: %v16f16u64 = call <16 x i64> @llvm.fptoui.sat.v16i64.v16f16(<16 x half> undef)
; AVX512F-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
;
@@ -1107,7 +1107,7 @@ define void @fp16() {
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 178 for instruction: %v16f16u16 = call <16 x i16> @llvm.fptoui.sat.v16i16.v16f16(<16 x half> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 178 for instruction: %v16f16s32 = call <16 x i32> @llvm.fptosi.sat.v16i32.v16f16(<16 x half> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 176 for instruction: %v16f16u32 = call <16 x i32> @llvm.fptoui.sat.v16i32.v16f16(<16 x half> undef)
-; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 186 for instruction: %v16f16s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f16(<16 x half> undef)
+; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 190 for instruction: %v16f16s64 = call <16 x i64> @llvm.fptosi.sat.v16i64.v16f16(<16 x half> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 183 for instruction: %v16f16u64 = call <16 x i64> @llvm.fptoui.sat.v16i64.v16f16(<16 x half> undef)
; AVX512DQ-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
;
diff --git a/llvm/test/Analysis/CostModel/X86/gather-scatter-mask-expansion.ll b/llvm/test/Analysis/CostModel/X86/gather-scatter-mask-expansion.ll
new file mode 100644
index 0000000000000..1e7553185c2fa
--- /dev/null
+++ b/llvm/test/Analysis/CostModel/X86/gather-scatter-mask-expansion.ll
@@ -0,0 +1,40 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+avx2 | FileCheck %s -check-prefixes=CHECK,AVX2
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+avx512f | FileCheck %s -check-prefixes=CHECK,AVX512
+
+; A masked gather/scatter whose data type legalizes into more register parts
+; than its <N x i1> mask needs extra kshiftr instructions on AVX-512 to extract
+; a sub-mask for each data part. That fanout is charged only on AVX-512, where
+; the mask lives in a single k-register; AVX2 keeps a full-width vector mask and
+; is not charged the extra term (its per-part gather cost already dominates).
+
+define <32 x i32> @gather_v32i32(<32 x ptr> %ptrs, <32 x i1> %m, <32 x i32> %pass) {
+; AVX2-LABEL: 'gather_v32i32'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 109 for instruction: %g = call <32 x i32> @llvm.masked.gather.v32i32.v32p0(<32 x ptr> align 4 %ptrs, <32 x i1> %m, <32 x i32> %pass)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <32 x i32> %g
+;
+; AVX512-LABEL: 'gather_v32i32'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 40 for instruction: %g = call <32 x i32> @llvm.masked.gather.v32i32.v32p0(<32 x ptr> align 4 %ptrs, <32 x i1> %m, <32 x i32> %pass)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <32 x i32> %g
+;
+ %g = call <32 x i32> @llvm.masked.gather.v32i32.v32p0(<32 x ptr> %ptrs, i32 4, <32 x i1> %m, <32 x i32> %pass)
+ ret <32 x i32> %g
+}
+
+define void @scatter_v32i32(<32 x i32> %v, <32 x ptr> %ptrs, <32 x i1> %m) {
+; AVX2-LABEL: 'scatter_v32i32'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 109 for instruction: call void @llvm.masked.scatter.v32i32.v32p0(<32 x i32> %v, <32 x ptr> align 4 %ptrs, <32 x i1> %m)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; AVX512-LABEL: 'scatter_v32i32'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 40 for instruction: call void @llvm.masked.scatter.v32i32.v32p0(<32 x i32> %v, <32 x ptr> align 4 %ptrs, <32 x i1> %m)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+ call void @llvm.masked.scatter.v32i32.v32p0(<32 x i32> %v, <32 x ptr> %ptrs, i32 4, <32 x i1> %m)
+ ret void
+}
+
+declare <32 x i32> @llvm.masked.gather.v32i32.v32p0(<32 x ptr>, i32, <32 x i1>, <32 x i32>)
+declare void @llvm.masked.scatter.v32i32.v32p0(<32 x i32>, <32 x ptr>, i32, <32 x i1>)
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK: {{.*}}
diff --git a/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost-inseltpoison.ll b/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost-inseltpoison.ll
index f835a8aedbbcc..4a950718fbd2e 100644
--- a/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost-inseltpoison.ll
+++ b/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost-inseltpoison.ll
@@ -128,22 +128,22 @@ define i32 @masked_load(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; SSE42-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 0
;
; AVX-LABEL: 'masked_load'
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V8F64 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V7F64 = call <7 x double> @llvm.masked.load.v7f64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V6F64 = call <6 x double> @llvm.masked.load.v6f64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V5F64 = call <5 x double> @llvm.masked.load.v5f64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V8F64 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V7F64 = call <7 x double> @llvm.masked.load.v7f64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V6F64 = call <6 x double> @llvm.masked.load.v6f64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V5F64 = call <5 x double> @llvm.masked.load.v5f64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x double> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V4F64 = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 1 undef, <4 x i1> %m4, <4 x double> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V3F64 = call <3 x double> @llvm.masked.load.v3f64.p0(ptr align 1 undef, <3 x i1> %m3, <3 x double> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V2F64 = call <2 x double> @llvm.masked.load.v2f64.p0(ptr align 1 undef, <2 x i1> %m2, <2 x double> undef)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1F64 = call <1 x double> @llvm.masked.load.v1f64.p0(ptr align 1 undef, <1 x i1> %m1, <1 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V16F32 = call <16 x float> @llvm.masked.load.v16f32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V15F32 = call <15 x float> @llvm.masked.load.v15f32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V14F32 = call <14 x float> @llvm.masked.load.v14f32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V13F32 = call <13 x float> @llvm.masked.load.v13f32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V12F32 = call <12 x float> @llvm.masked.load.v12f32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V11F32 = call <11 x float> @llvm.masked.load.v11f32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V10F32 = call <10 x float> @llvm.masked.load.v10f32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V9F32 = call <9 x float> @llvm.masked.load.v9f32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V16F32 = call <16 x float> @llvm.masked.load.v16f32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V15F32 = call <15 x float> @llvm.masked.load.v15f32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V14F32 = call <14 x float> @llvm.masked.load.v14f32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V13F32 = call <13 x float> @llvm.masked.load.v13f32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V12F32 = call <12 x float> @llvm.masked.load.v12f32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V11F32 = call <11 x float> @llvm.masked.load.v11f32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V10F32 = call <10 x float> @llvm.masked.load.v10f32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V9F32 = call <9 x float> @llvm.masked.load.v9f32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V8F32 = call <8 x float> @llvm.masked.load.v8f32.p0(ptr align 1 undef, <8 x i1> %m8, <8 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V7F32 = call <7 x float> @llvm.masked.load.v7f32.p0(ptr align 1 undef, <7 x i1> %m7, <7 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V6F32 = call <6 x float> @llvm.masked.load.v6f32.p0(ptr align 1 undef, <6 x i1> %m6, <6 x float> undef)
@@ -152,22 +152,22 @@ define i32 @masked_load(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; AVX-NEXT: Cost Model: Found costs of 3 for: %V3F32 = call <3 x float> @llvm.masked.load.v3f32.p0(ptr align 1 undef, <3 x i1> %m3, <3 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V2F32 = call <2 x float> @llvm.masked.load.v2f32.p0(ptr align 1 undef, <2 x i1> %m2, <2 x float> undef)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1F32 = call <1 x float> @llvm.masked.load.v1f32.p0(ptr align 1 undef, <1 x i1> %m1, <1 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V8I64 = call <8 x i64> @llvm.masked.load.v8i64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V7I64 = call <7 x i64> @llvm.masked.load.v7i64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V6I64 = call <6 x i64> @llvm.masked.load.v6i64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V5I64 = call <5 x i64> @llvm.masked.load.v5i64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V8I64 = call <8 x i64> @llvm.masked.load.v8i64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V7I64 = call <7 x i64> @llvm.masked.load.v7i64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V6I64 = call <6 x i64> @llvm.masked.load.v6i64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V5I64 = call <5 x i64> @llvm.masked.load.v5i64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V4I64 = call <4 x i64> @llvm.masked.load.v4i64.p0(ptr align 1 undef, <4 x i1> %m4, <4 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V3I64 = call <3 x i64> @llvm.masked.load.v3i64.p0(ptr align 1 undef, <3 x i1> %m3, <3 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V2I64 = call <2 x i64> @llvm.masked.load.v2i64.p0(ptr align 1 undef, <2 x i1> %m2, <2 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1I64 = call <1 x i64> @llvm.masked.load.v1i64.p0(ptr align 1 undef, <1 x i1> %m1, <1 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V16I32 = call <16 x i32> @llvm.masked.load.v16i32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V15I32 = call <15 x i32> @llvm.masked.load.v15i32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V14I32 = call <14 x i32> @llvm.masked.load.v14i32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V13I32 = call <13 x i32> @llvm.masked.load.v13i32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V12I32 = call <12 x i32> @llvm.masked.load.v12i32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V11I32 = call <11 x i32> @llvm.masked.load.v11i32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V10I32 = call <10 x i32> @llvm.masked.load.v10i32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V9I32 = call <9 x i32> @llvm.masked.load.v9i32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V16I32 = call <16 x i32> @llvm.masked.load.v16i32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V15I32 = call <15 x i32> @llvm.masked.load.v15i32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V14I32 = call <14 x i32> @llvm.masked.load.v14i32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V13I32 = call <13 x i32> @llvm.masked.load.v13i32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V12I32 = call <12 x i32> @llvm.masked.load.v12i32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V11I32 = call <11 x i32> @llvm.masked.load.v11i32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V10I32 = call <10 x i32> @llvm.masked.load.v10i32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V9I32 = call <9 x i32> @llvm.masked.load.v9i32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x i32> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V8I32 = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 undef, <8 x i1> %m8, <8 x i32> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V7I32 = call <7 x i32> @llvm.masked.load.v7i32.p0(ptr align 1 undef, <7 x i1> %m7, <7 x i32> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V6I32 = call <6 x i32> @llvm.masked.load.v6i32.p0(ptr align 1 undef, <6 x i1> %m6, <6 x i32> undef)
@@ -489,22 +489,22 @@ define i32 @masked_store(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; SSE42-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 0
;
; AVX-LABEL: 'masked_store'
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v8f64.p0(<8 x double> undef, ptr align 1 undef, <8 x i1> %m8)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v7f64.p0(<7 x double> undef, ptr align 1 undef, <7 x i1> %m7)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v6f64.p0(<6 x double> undef, ptr align 1 undef, <6 x i1> %m6)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v5f64.p0(<5 x double> undef, ptr align 1 undef, <5 x i1> %m5)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v8f64.p0(<8 x double> undef, ptr align 1 undef, <8 x i1> %m8)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v7f64.p0(<7 x double> undef, ptr align 1 undef, <7 x i1> %m7)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v6f64.p0(<6 x double> undef, ptr align 1 undef, <6 x i1> %m6)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v5f64.p0(<5 x double> undef, ptr align 1 undef, <5 x i1> %m5)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v4f64.p0(<4 x double> undef, ptr align 1 undef, <4 x i1> %m4)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v3f64.p0(<3 x double> undef, ptr align 1 undef, <3 x i1> %m3)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v2f64.p0(<2 x double> undef, ptr align 1 undef, <2 x i1> %m2)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:3 SizeLat:3 for: call void @llvm.masked.store.v1f64.p0(<1 x double> undef, ptr align 1 undef, <1 x i1> %m1)
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v16f32.p0(<16 x float> undef, ptr align 1 undef, <16 x i1> %m16)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v15f32.p0(<15 x float> undef, ptr align 1 undef, <15 x i1> %m15)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v14f32.p0(<14 x float> undef, ptr align 1 undef, <14 x i1> %m14)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v13f32.p0(<13 x float> undef, ptr align 1 undef, <13 x i1> %m13)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v12f32.p0(<12 x float> undef, ptr align 1 undef, <12 x i1> %m12)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v11f32.p0(<11 x float> undef, ptr align 1 undef, <11 x i1> %m11)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v10f32.p0(<10 x float> undef, ptr align 1 undef, <10 x i1> %m10)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v9f32.p0(<9 x float> undef, ptr align 1 undef, <9 x i1> %m9)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v16f32.p0(<16 x float> undef, ptr align 1 undef, <16 x i1> %m16)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v15f32.p0(<15 x float> undef, ptr align 1 undef, <15 x i1> %m15)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v14f32.p0(<14 x float> undef, ptr align 1 undef, <14 x i1> %m14)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v13f32.p0(<13 x float> undef, ptr align 1 undef, <13 x i1> %m13)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v12f32.p0(<12 x float> undef, ptr align 1 undef, <12 x i1> %m12)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v11f32.p0(<11 x float> undef, ptr align 1 undef, <11 x i1> %m11)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v10f32.p0(<10 x float> undef, ptr align 1 undef, <10 x i1> %m10)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v9f32.p0(<9 x float> undef, ptr align 1 undef, <9 x i1> %m9)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v8f32.p0(<8 x float> undef, ptr align 1 undef, <8 x i1> %m8)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v7f32.p0(<7 x float> undef, ptr align 1 undef, <7 x i1> %m7)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v6f32.p0(<6 x float> undef, ptr align 1 undef, <6 x i1> %m6)
@@ -513,22 +513,22 @@ define i32 @masked_store(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v3f32.p0(<3 x float> undef, ptr align 1 undef, <3 x i1> %m3)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v2f32.p0(<2 x float> undef, ptr align 1 undef, <2 x i1> %m2)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:3 SizeLat:3 for: call void @llvm.masked.store.v1f32.p0(<1 x float> undef, ptr align 1 undef, <1 x i1> %m1)
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v8i64.p0(<8 x i64> undef, ptr align 1 undef, <8 x i1> %m8)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v7i64.p0(<7 x i64> undef, ptr align 1 undef, <7 x i1> %m7)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v6i64.p0(<6 x i64> undef, ptr align 1 undef, <6 x i1> %m6)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v5i64.p0(<5 x i64> undef, ptr align 1 undef, <5 x i1> %m5)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v8i64.p0(<8 x i64> undef, ptr align 1 undef, <8 x i1> %m8)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v7i64.p0(<7 x i64> undef, ptr align 1 undef, <7 x i1> %m7)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v6i64.p0(<6 x i64> undef, ptr align 1 undef, <6 x i1> %m6)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v5i64.p0(<5 x i64> undef, ptr align 1 undef, <5 x i1> %m5)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v4i64.p0(<4 x i64> undef, ptr align 1 undef, <4 x i1> %m4)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v3i64.p0(<3 x i64> undef, ptr align 1 undef, <3 x i1> %m3)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v2i64.p0(<2 x i64> undef, ptr align 1 undef, <2 x i1> %m2)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:3 SizeLat:3 for: call void @llvm.masked.store.v1i64.p0(<1 x i64> undef, ptr align 1 undef, <1 x i1> %m1)
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v16i32.p0(<16 x i32> undef, ptr align 1 undef, <16 x i1> %m16)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v15i32.p0(<15 x i32> undef, ptr align 1 undef, <15 x i1> %m15)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v14i32.p0(<14 x i32> undef, ptr align 1 undef, <14 x i1> %m14)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v13i32.p0(<13 x i32> undef, ptr align 1 undef, <13 x i1> %m13)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v12i32.p0(<12 x i32> undef, ptr align 1 undef, <12 x i1> %m12)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v11i32.p0(<11 x i32> undef, ptr align 1 undef, <11 x i1> %m11)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v10i32.p0(<10 x i32> undef, ptr align 1 undef, <10 x i1> %m10)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v9i32.p0(<9 x i32> undef, ptr align 1 undef, <9 x i1> %m9)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v16i32.p0(<16 x i32> undef, ptr align 1 undef, <16 x i1> %m16)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v15i32.p0(<15 x i32> undef, ptr align 1 undef, <15 x i1> %m15)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v14i32.p0(<14 x i32> undef, ptr align 1 undef, <14 x i1> %m14)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v13i32.p0(<13 x i32> undef, ptr align 1 undef, <13 x i1> %m13)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v12i32.p0(<12 x i32> undef, ptr align 1 undef, <12 x i1> %m12)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v11i32.p0(<11 x i32> undef, ptr align 1 undef, <11 x i1> %m11)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v10i32.p0(<10 x i32> undef, ptr align 1 undef, <10 x i1> %m10)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v9i32.p0(<9 x i32> undef, ptr align 1 undef, <9 x i1> %m9)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v8i32.p0(<8 x i32> undef, ptr align 1 undef, <8 x i1> %m8)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v7i32.p0(<7 x i32> undef, ptr align 1 undef, <7 x i1> %m7)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v6i32.p0(<6 x i32> undef, ptr align 1 undef, <6 x i1> %m6)
@@ -840,19 +840,19 @@ define i32 @masked_gather(<1 x i1> %m1, <2 x i1> %m2, <4 x i1> %m4, <8 x i1> %m8
; AVX2-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 0
;
; SKL-LABEL: 'masked_gather'
-; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8F64 = call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x double> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:16 CodeSize:6 Lat:40 SizeLat:16 for: %V8F64 = call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x double> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4F64 = call <4 x double> @llvm.masked.gather.v4f64.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x double> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2F64 = call <2 x double> @llvm.masked.gather.v2f64.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x double> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1F64 = call <1 x double> @llvm.masked.gather.v1f64.v1p0(<1 x ptr> align 1 undef, <1 x i1> %m1, <1 x double> undef)
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %V16F32 = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %V16F32 = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8F32 = call <8 x float> @llvm.masked.gather.v8f32.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4F32 = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2F32 = call <2 x float> @llvm.masked.gather.v2f32.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x float> undef)
-; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8I64 = call <8 x i64> @llvm.masked.gather.v8i64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x i64> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:16 CodeSize:6 Lat:40 SizeLat:16 for: %V8I64 = call <8 x i64> @llvm.masked.gather.v8i64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x i64> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4I64 = call <4 x i64> @llvm.masked.gather.v4i64.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x i64> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2I64 = call <2 x i64> @llvm.masked.gather.v2i64.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x i64> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1I64 = call <1 x i64> @llvm.masked.gather.v1i64.v1p0(<1 x ptr> align 1 undef, <1 x i1> %m1, <1 x i64> undef)
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %V16I32 = call <16 x i32> @llvm.masked.gather.v16i32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x i32> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %V16I32 = call <16 x i32> @llvm.masked.gather.v16i32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x i32> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8I32 = call <8 x i32> @llvm.masked.gather.v8i32.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x i32> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4I32 = call <4 x i32> @llvm.masked.gather.v4i32.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x i32> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2I32 = call <2 x i32> @llvm.masked.gather.v2i32.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x i32> undef)
@@ -1909,7 +1909,7 @@ define <16 x float> @test_gather_16f32_const_mask(ptr %base, <16 x i32> %ind) {
; SKL-LABEL: 'test_gather_16f32_const_mask'
; SKL-NEXT: Cost Model: Found costs of RThru:10 CodeSize:1 Lat:1 SizeLat:1 for: %sext_ind = sext <16 x i32> %ind to <16 x i64>
; SKL-NEXT: Cost Model: Found costs of 0 for: %gep.v = getelementptr float, ptr %base, <16 x i64> %sext_ind
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> splat (i1 true), <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> splat (i1 true), <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret <16 x float> %res
;
; AVX512-LABEL: 'test_gather_16f32_const_mask'
@@ -1953,7 +1953,7 @@ define <16 x float> @test_gather_16f32_var_mask(ptr %base, <16 x i32> %ind, <16
; SKL-LABEL: 'test_gather_16f32_var_mask'
; SKL-NEXT: Cost Model: Found costs of RThru:10 CodeSize:1 Lat:1 SizeLat:1 for: %sext_ind = sext <16 x i32> %ind to <16 x i64>
; SKL-NEXT: Cost Model: Found costs of 0 for: %gep.v = getelementptr float, ptr %base, <16 x i64> %sext_ind
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret <16 x float> %res
;
; AVX512-LABEL: 'test_gather_16f32_var_mask'
@@ -1997,7 +1997,7 @@ define <16 x float> @test_gather_16f32_ra_var_mask(<16 x ptr> %ptrs, <16 x i32>
; SKL-LABEL: 'test_gather_16f32_ra_var_mask'
; SKL-NEXT: Cost Model: Found costs of RThru:10 CodeSize:1 Lat:1 SizeLat:1 for: %sext_ind = sext <16 x i32> %ind to <16 x i64>
; SKL-NEXT: Cost Model: Found costs of 0 for: %gep.v = getelementptr float, <16 x ptr> %ptrs, <16 x i64> %sext_ind
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret <16 x float> %res
;
; AVX512-LABEL: 'test_gather_16f32_ra_var_mask'
@@ -2051,7 +2051,7 @@ define <16 x float> @test_gather_16f32_const_mask2(ptr %base, <16 x i32> %ind) {
; SKL-NEXT: Cost Model: Found costs of RThru:1 CodeSize:1 Lat:3 SizeLat:2 for: %broadcast.splat = shufflevector <16 x ptr> %broadcast.splatinsert, <16 x ptr> poison, <16 x i32> zeroinitializer
; SKL-NEXT: Cost Model: Found costs of RThru:10 CodeSize:1 Lat:1 SizeLat:1 for: %sext_ind = sext <16 x i32> %ind to <16 x i64>
; SKL-NEXT: Cost Model: Found costs of 0 for: %gep.random = getelementptr float, <16 x ptr> %broadcast.splat, <16 x i64> %sext_ind
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.random, <16 x i1> splat (i1 true), <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.random, <16 x i1> splat (i1 true), <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret <16 x float> %res
;
; AVX512-LABEL: 'test_gather_16f32_const_mask2'
diff --git a/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost.ll b/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost.ll
index ed1b534fac8f8..5b4d313eee7c3 100644
--- a/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost.ll
+++ b/llvm/test/Analysis/CostModel/X86/masked-intrinsic-cost.ll
@@ -128,22 +128,22 @@ define i32 @masked_load(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; SSE42-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 0
;
; AVX-LABEL: 'masked_load'
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V8F64 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V7F64 = call <7 x double> @llvm.masked.load.v7f64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V6F64 = call <6 x double> @llvm.masked.load.v6f64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V5F64 = call <5 x double> @llvm.masked.load.v5f64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V8F64 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V7F64 = call <7 x double> @llvm.masked.load.v7f64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V6F64 = call <6 x double> @llvm.masked.load.v6f64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x double> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V5F64 = call <5 x double> @llvm.masked.load.v5f64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x double> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V4F64 = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 1 undef, <4 x i1> %m4, <4 x double> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V3F64 = call <3 x double> @llvm.masked.load.v3f64.p0(ptr align 1 undef, <3 x i1> %m3, <3 x double> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V2F64 = call <2 x double> @llvm.masked.load.v2f64.p0(ptr align 1 undef, <2 x i1> %m2, <2 x double> undef)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1F64 = call <1 x double> @llvm.masked.load.v1f64.p0(ptr align 1 undef, <1 x i1> %m1, <1 x double> undef)
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V16F32 = call <16 x float> @llvm.masked.load.v16f32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V15F32 = call <15 x float> @llvm.masked.load.v15f32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V14F32 = call <14 x float> @llvm.masked.load.v14f32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V13F32 = call <13 x float> @llvm.masked.load.v13f32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V12F32 = call <12 x float> @llvm.masked.load.v12f32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V11F32 = call <11 x float> @llvm.masked.load.v11f32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V10F32 = call <10 x float> @llvm.masked.load.v10f32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V9F32 = call <9 x float> @llvm.masked.load.v9f32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V16F32 = call <16 x float> @llvm.masked.load.v16f32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V15F32 = call <15 x float> @llvm.masked.load.v15f32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V14F32 = call <14 x float> @llvm.masked.load.v14f32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V13F32 = call <13 x float> @llvm.masked.load.v13f32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V12F32 = call <12 x float> @llvm.masked.load.v12f32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V11F32 = call <11 x float> @llvm.masked.load.v11f32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V10F32 = call <10 x float> @llvm.masked.load.v10f32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x float> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V9F32 = call <9 x float> @llvm.masked.load.v9f32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V8F32 = call <8 x float> @llvm.masked.load.v8f32.p0(ptr align 1 undef, <8 x i1> %m8, <8 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V7F32 = call <7 x float> @llvm.masked.load.v7f32.p0(ptr align 1 undef, <7 x i1> %m7, <7 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V6F32 = call <6 x float> @llvm.masked.load.v6f32.p0(ptr align 1 undef, <6 x i1> %m6, <6 x float> undef)
@@ -152,22 +152,22 @@ define i32 @masked_load(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; AVX-NEXT: Cost Model: Found costs of 3 for: %V3F32 = call <3 x float> @llvm.masked.load.v3f32.p0(ptr align 1 undef, <3 x i1> %m3, <3 x float> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V2F32 = call <2 x float> @llvm.masked.load.v2f32.p0(ptr align 1 undef, <2 x i1> %m2, <2 x float> undef)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1F32 = call <1 x float> @llvm.masked.load.v1f32.p0(ptr align 1 undef, <1 x i1> %m1, <1 x float> undef)
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V8I64 = call <8 x i64> @llvm.masked.load.v8i64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V7I64 = call <7 x i64> @llvm.masked.load.v7i64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V6I64 = call <6 x i64> @llvm.masked.load.v6i64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V5I64 = call <5 x i64> @llvm.masked.load.v5i64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V8I64 = call <8 x i64> @llvm.masked.load.v8i64.p0(ptr align 1 undef, <8 x i1> %m8, <8 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V7I64 = call <7 x i64> @llvm.masked.load.v7i64.p0(ptr align 1 undef, <7 x i1> %m7, <7 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V6I64 = call <6 x i64> @llvm.masked.load.v6i64.p0(ptr align 1 undef, <6 x i1> %m6, <6 x i64> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V5I64 = call <5 x i64> @llvm.masked.load.v5i64.p0(ptr align 1 undef, <5 x i1> %m5, <5 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V4I64 = call <4 x i64> @llvm.masked.load.v4i64.p0(ptr align 1 undef, <4 x i1> %m4, <4 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V3I64 = call <3 x i64> @llvm.masked.load.v3i64.p0(ptr align 1 undef, <3 x i1> %m3, <3 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V2I64 = call <2 x i64> @llvm.masked.load.v2i64.p0(ptr align 1 undef, <2 x i1> %m2, <2 x i64> undef)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1I64 = call <1 x i64> @llvm.masked.load.v1i64.p0(ptr align 1 undef, <1 x i1> %m1, <1 x i64> undef)
-; AVX-NEXT: Cost Model: Found costs of 4 for: %V16I32 = call <16 x i32> @llvm.masked.load.v16i32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V15I32 = call <15 x i32> @llvm.masked.load.v15i32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V14I32 = call <14 x i32> @llvm.masked.load.v14i32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V13I32 = call <13 x i32> @llvm.masked.load.v13i32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V12I32 = call <12 x i32> @llvm.masked.load.v12i32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V11I32 = call <11 x i32> @llvm.masked.load.v11i32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V10I32 = call <10 x i32> @llvm.masked.load.v10i32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x i32> undef)
-; AVX-NEXT: Cost Model: Found costs of 5 for: %V9I32 = call <9 x i32> @llvm.masked.load.v9i32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 8 for: %V16I32 = call <16 x i32> @llvm.masked.load.v16i32.p0(ptr align 1 undef, <16 x i1> %m16, <16 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V15I32 = call <15 x i32> @llvm.masked.load.v15i32.p0(ptr align 1 undef, <15 x i1> %m15, <15 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V14I32 = call <14 x i32> @llvm.masked.load.v14i32.p0(ptr align 1 undef, <14 x i1> %m14, <14 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V13I32 = call <13 x i32> @llvm.masked.load.v13i32.p0(ptr align 1 undef, <13 x i1> %m13, <13 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V12I32 = call <12 x i32> @llvm.masked.load.v12i32.p0(ptr align 1 undef, <12 x i1> %m12, <12 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V11I32 = call <11 x i32> @llvm.masked.load.v11i32.p0(ptr align 1 undef, <11 x i1> %m11, <11 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V10I32 = call <10 x i32> @llvm.masked.load.v10i32.p0(ptr align 1 undef, <10 x i1> %m10, <10 x i32> undef)
+; AVX-NEXT: Cost Model: Found costs of 9 for: %V9I32 = call <9 x i32> @llvm.masked.load.v9i32.p0(ptr align 1 undef, <9 x i1> %m9, <9 x i32> undef)
; AVX-NEXT: Cost Model: Found costs of 2 for: %V8I32 = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 undef, <8 x i1> %m8, <8 x i32> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V7I32 = call <7 x i32> @llvm.masked.load.v7i32.p0(ptr align 1 undef, <7 x i1> %m7, <7 x i32> undef)
; AVX-NEXT: Cost Model: Found costs of 3 for: %V6I32 = call <6 x i32> @llvm.masked.load.v6i32.p0(ptr align 1 undef, <6 x i1> %m6, <6 x i32> undef)
@@ -489,22 +489,22 @@ define i32 @masked_store(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; SSE42-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 0
;
; AVX-LABEL: 'masked_store'
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v8f64.p0(<8 x double> undef, ptr align 1 undef, <8 x i1> %m8)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v7f64.p0(<7 x double> undef, ptr align 1 undef, <7 x i1> %m7)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v6f64.p0(<6 x double> undef, ptr align 1 undef, <6 x i1> %m6)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v5f64.p0(<5 x double> undef, ptr align 1 undef, <5 x i1> %m5)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v8f64.p0(<8 x double> undef, ptr align 1 undef, <8 x i1> %m8)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v7f64.p0(<7 x double> undef, ptr align 1 undef, <7 x i1> %m7)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v6f64.p0(<6 x double> undef, ptr align 1 undef, <6 x i1> %m6)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v5f64.p0(<5 x double> undef, ptr align 1 undef, <5 x i1> %m5)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v4f64.p0(<4 x double> undef, ptr align 1 undef, <4 x i1> %m4)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v3f64.p0(<3 x double> undef, ptr align 1 undef, <3 x i1> %m3)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v2f64.p0(<2 x double> undef, ptr align 1 undef, <2 x i1> %m2)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:3 SizeLat:3 for: call void @llvm.masked.store.v1f64.p0(<1 x double> undef, ptr align 1 undef, <1 x i1> %m1)
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v16f32.p0(<16 x float> undef, ptr align 1 undef, <16 x i1> %m16)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v15f32.p0(<15 x float> undef, ptr align 1 undef, <15 x i1> %m15)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v14f32.p0(<14 x float> undef, ptr align 1 undef, <14 x i1> %m14)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v13f32.p0(<13 x float> undef, ptr align 1 undef, <13 x i1> %m13)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v12f32.p0(<12 x float> undef, ptr align 1 undef, <12 x i1> %m12)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v11f32.p0(<11 x float> undef, ptr align 1 undef, <11 x i1> %m11)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v10f32.p0(<10 x float> undef, ptr align 1 undef, <10 x i1> %m10)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v9f32.p0(<9 x float> undef, ptr align 1 undef, <9 x i1> %m9)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v16f32.p0(<16 x float> undef, ptr align 1 undef, <16 x i1> %m16)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v15f32.p0(<15 x float> undef, ptr align 1 undef, <15 x i1> %m15)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v14f32.p0(<14 x float> undef, ptr align 1 undef, <14 x i1> %m14)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v13f32.p0(<13 x float> undef, ptr align 1 undef, <13 x i1> %m13)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v12f32.p0(<12 x float> undef, ptr align 1 undef, <12 x i1> %m12)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v11f32.p0(<11 x float> undef, ptr align 1 undef, <11 x i1> %m11)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v10f32.p0(<10 x float> undef, ptr align 1 undef, <10 x i1> %m10)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v9f32.p0(<9 x float> undef, ptr align 1 undef, <9 x i1> %m9)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v8f32.p0(<8 x float> undef, ptr align 1 undef, <8 x i1> %m8)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v7f32.p0(<7 x float> undef, ptr align 1 undef, <7 x i1> %m7)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v6f32.p0(<6 x float> undef, ptr align 1 undef, <6 x i1> %m6)
@@ -513,22 +513,22 @@ define i32 @masked_store(<1 x i1> %m1, <2 x i1> %m2, <3 x i1> %m3, <4 x i1> %m4,
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v3f32.p0(<3 x float> undef, ptr align 1 undef, <3 x i1> %m3)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v2f32.p0(<2 x float> undef, ptr align 1 undef, <2 x i1> %m2)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:3 SizeLat:3 for: call void @llvm.masked.store.v1f32.p0(<1 x float> undef, ptr align 1 undef, <1 x i1> %m1)
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v8i64.p0(<8 x i64> undef, ptr align 1 undef, <8 x i1> %m8)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v7i64.p0(<7 x i64> undef, ptr align 1 undef, <7 x i1> %m7)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v6i64.p0(<6 x i64> undef, ptr align 1 undef, <6 x i1> %m6)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v5i64.p0(<5 x i64> undef, ptr align 1 undef, <5 x i1> %m5)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v8i64.p0(<8 x i64> undef, ptr align 1 undef, <8 x i1> %m8)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v7i64.p0(<7 x i64> undef, ptr align 1 undef, <7 x i1> %m7)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v6i64.p0(<6 x i64> undef, ptr align 1 undef, <6 x i1> %m6)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v5i64.p0(<5 x i64> undef, ptr align 1 undef, <5 x i1> %m5)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v4i64.p0(<4 x i64> undef, ptr align 1 undef, <4 x i1> %m4)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v3i64.p0(<3 x i64> undef, ptr align 1 undef, <3 x i1> %m3)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v2i64.p0(<2 x i64> undef, ptr align 1 undef, <2 x i1> %m2)
; AVX-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:3 SizeLat:3 for: call void @llvm.masked.store.v1i64.p0(<1 x i64> undef, ptr align 1 undef, <1 x i1> %m1)
-; AVX-NEXT: Cost Model: Found costs of 16 for: call void @llvm.masked.store.v16i32.p0(<16 x i32> undef, ptr align 1 undef, <16 x i1> %m16)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v15i32.p0(<15 x i32> undef, ptr align 1 undef, <15 x i1> %m15)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v14i32.p0(<14 x i32> undef, ptr align 1 undef, <14 x i1> %m14)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v13i32.p0(<13 x i32> undef, ptr align 1 undef, <13 x i1> %m13)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v12i32.p0(<12 x i32> undef, ptr align 1 undef, <12 x i1> %m12)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v11i32.p0(<11 x i32> undef, ptr align 1 undef, <11 x i1> %m11)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v10i32.p0(<10 x i32> undef, ptr align 1 undef, <10 x i1> %m10)
-; AVX-NEXT: Cost Model: Found costs of 17 for: call void @llvm.masked.store.v9i32.p0(<9 x i32> undef, ptr align 1 undef, <9 x i1> %m9)
+; AVX-NEXT: Cost Model: Found costs of 20 for: call void @llvm.masked.store.v16i32.p0(<16 x i32> undef, ptr align 1 undef, <16 x i1> %m16)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v15i32.p0(<15 x i32> undef, ptr align 1 undef, <15 x i1> %m15)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v14i32.p0(<14 x i32> undef, ptr align 1 undef, <14 x i1> %m14)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v13i32.p0(<13 x i32> undef, ptr align 1 undef, <13 x i1> %m13)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v12i32.p0(<12 x i32> undef, ptr align 1 undef, <12 x i1> %m12)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v11i32.p0(<11 x i32> undef, ptr align 1 undef, <11 x i1> %m11)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v10i32.p0(<10 x i32> undef, ptr align 1 undef, <10 x i1> %m10)
+; AVX-NEXT: Cost Model: Found costs of 21 for: call void @llvm.masked.store.v9i32.p0(<9 x i32> undef, ptr align 1 undef, <9 x i1> %m9)
; AVX-NEXT: Cost Model: Found costs of 8 for: call void @llvm.masked.store.v8i32.p0(<8 x i32> undef, ptr align 1 undef, <8 x i1> %m8)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v7i32.p0(<7 x i32> undef, ptr align 1 undef, <7 x i1> %m7)
; AVX-NEXT: Cost Model: Found costs of 9 for: call void @llvm.masked.store.v6i32.p0(<6 x i32> undef, ptr align 1 undef, <6 x i1> %m6)
@@ -840,19 +840,19 @@ define i32 @masked_gather(<1 x i1> %m1, <2 x i1> %m2, <4 x i1> %m4, <8 x i1> %m8
; AVX2-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret i32 0
;
; SKL-LABEL: 'masked_gather'
-; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8F64 = call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x double> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:16 CodeSize:6 Lat:40 SizeLat:16 for: %V8F64 = call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x double> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4F64 = call <4 x double> @llvm.masked.gather.v4f64.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x double> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2F64 = call <2 x double> @llvm.masked.gather.v2f64.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x double> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1F64 = call <1 x double> @llvm.masked.gather.v1f64.v1p0(<1 x ptr> align 1 undef, <1 x i1> %m1, <1 x double> undef)
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %V16F32 = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %V16F32 = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8F32 = call <8 x float> @llvm.masked.gather.v8f32.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4F32 = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2F32 = call <2 x float> @llvm.masked.gather.v2f32.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x float> undef)
-; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8I64 = call <8 x i64> @llvm.masked.gather.v8i64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x i64> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:16 CodeSize:6 Lat:40 SizeLat:16 for: %V8I64 = call <8 x i64> @llvm.masked.gather.v8i64.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x i64> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4I64 = call <4 x i64> @llvm.masked.gather.v4i64.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x i64> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2I64 = call <2 x i64> @llvm.masked.gather.v2i64.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x i64> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:2 CodeSize:3 Lat:6 SizeLat:3 for: %V1I64 = call <1 x i64> @llvm.masked.gather.v1i64.v1p0(<1 x ptr> align 1 undef, <1 x i1> %m1, <1 x i64> undef)
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %V16I32 = call <16 x i32> @llvm.masked.gather.v16i32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x i32> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %V16I32 = call <16 x i32> @llvm.masked.gather.v16i32.v16p0(<16 x ptr> align 1 undef, <16 x i1> %m16, <16 x i32> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:12 CodeSize:2 Lat:36 SizeLat:12 for: %V8I32 = call <8 x i32> @llvm.masked.gather.v8i32.v8p0(<8 x ptr> align 1 undef, <8 x i1> %m8, <8 x i32> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:6 CodeSize:1 Lat:18 SizeLat:6 for: %V4I32 = call <4 x i32> @llvm.masked.gather.v4i32.v4p0(<4 x ptr> align 1 undef, <4 x i1> %m4, <4 x i32> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:4 CodeSize:1 Lat:10 SizeLat:4 for: %V2I32 = call <2 x i32> @llvm.masked.gather.v2i32.v2p0(<2 x ptr> align 1 undef, <2 x i1> %m2, <2 x i32> undef)
@@ -1909,7 +1909,7 @@ define <16 x float> @test_gather_16f32_const_mask(ptr %base, <16 x i32> %ind) {
; SKL-LABEL: 'test_gather_16f32_const_mask'
; SKL-NEXT: Cost Model: Found costs of RThru:10 CodeSize:1 Lat:1 SizeLat:1 for: %sext_ind = sext <16 x i32> %ind to <16 x i64>
; SKL-NEXT: Cost Model: Found costs of 0 for: %gep.v = getelementptr float, ptr %base, <16 x i64> %sext_ind
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> splat (i1 true), <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> splat (i1 true), <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret <16 x float> %res
;
; AVX512-LABEL: 'test_gather_16f32_const_mask'
@@ -1953,7 +1953,7 @@ define <16 x float> @test_gather_16f32_var_mask(ptr %base, <16 x i32> %ind, <16
; SKL-LABEL: 'test_gather_16f32_var_mask'
; SKL-NEXT: Cost Model: Found costs of RThru:10 CodeSize:1 Lat:1 SizeLat:1 for: %sext_ind = sext <16 x i32> %ind to <16 x i64>
; SKL-NEXT: Cost Model: Found costs of 0 for: %gep.v = getelementptr float, ptr %base, <16 x i64> %sext_ind
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret <16 x float> %res
;
; AVX512-LABEL: 'test_gather_16f32_var_mask'
@@ -1997,7 +1997,7 @@ define <16 x float> @test_gather_16f32_ra_var_mask(<16 x ptr> %ptrs, <16 x i32>
; SKL-LABEL: 'test_gather_16f32_ra_var_mask'
; SKL-NEXT: Cost Model: Found costs of RThru:10 CodeSize:1 Lat:1 SizeLat:1 for: %sext_ind = sext <16 x i32> %ind to <16 x i64>
; SKL-NEXT: Cost Model: Found costs of 0 for: %gep.v = getelementptr float, <16 x ptr> %ptrs, <16 x i64> %sext_ind
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.v, <16 x i1> %mask, <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret <16 x float> %res
;
; AVX512-LABEL: 'test_gather_16f32_ra_var_mask'
@@ -2051,7 +2051,7 @@ define <16 x float> @test_gather_16f32_const_mask2(ptr %base, <16 x i32> %ind) {
; SKL-NEXT: Cost Model: Found costs of RThru:1 CodeSize:1 Lat:3 SizeLat:2 for: %broadcast.splat = shufflevector <16 x ptr> %broadcast.splatinsert, <16 x ptr> undef, <16 x i32> zeroinitializer
; SKL-NEXT: Cost Model: Found costs of RThru:10 CodeSize:1 Lat:1 SizeLat:1 for: %sext_ind = sext <16 x i32> %ind to <16 x i64>
; SKL-NEXT: Cost Model: Found costs of 0 for: %gep.random = getelementptr float, <16 x ptr> %broadcast.splat, <16 x i64> %sext_ind
-; SKL-NEXT: Cost Model: Found costs of RThru:24 CodeSize:4 Lat:72 SizeLat:24 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.random, <16 x i1> splat (i1 true), <16 x float> undef)
+; SKL-NEXT: Cost Model: Found costs of RThru:28 CodeSize:8 Lat:76 SizeLat:28 for: %res = call <16 x float> @llvm.masked.gather.v16f32.v16p0(<16 x ptr> align 4 %gep.random, <16 x i1> splat (i1 true), <16 x float> undef)
; SKL-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret <16 x float> %res
;
; AVX512-LABEL: 'test_gather_16f32_const_mask2'
diff --git a/llvm/test/Analysis/CostModel/X86/masked-mem-mask-expansion.ll b/llvm/test/Analysis/CostModel/X86/masked-mem-mask-expansion.ll
new file mode 100644
index 0000000000000..6901f2beacd0b
--- /dev/null
+++ b/llvm/test/Analysis/CostModel/X86/masked-mem-mask-expansion.ll
@@ -0,0 +1,89 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+sse2 | FileCheck %s -check-prefixes=SSE2
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+avx2 | FileCheck %s -check-prefixes=AVX2
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+avx512f | FileCheck %s -check-prefixes=AVX512
+
+; Predicate mask-expansion (fanout) cost for masked memory ops. The mask is a
+; compact <N x i1>; once the data type splits into more register parts than the
+; mask, each extra part needs its own sub-mask materialized, so the per-lane
+; cost must rise with the width (it is flat without the fanout charge). This is
+; charged on every subtarget where the masked op is legal, not just AVX-512.
+
+define void @masked_store_f64(ptr %p) {
+; SSE2-LABEL: 'masked_store_f64'
+; SSE2-NEXT: Cost Model: Found an estimated cost of 17 for instruction: call void @llvm.masked.store.v4f64.p0(<4 x double> undef, ptr align 8 %p, <4 x i1> undef)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 35 for instruction: call void @llvm.masked.store.v8f64.p0(<8 x double> undef, ptr align 8 %p, <8 x i1> undef)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 71 for instruction: call void @llvm.masked.store.v16f64.p0(<16 x double> undef, ptr align 8 %p, <16 x i1> undef)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 142 for instruction: call void @llvm.masked.store.v32f64.p0(<32 x double> undef, ptr align 8 %p, <32 x i1> undef)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; AVX2-LABEL: 'masked_store_f64'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 8 for instruction: call void @llvm.masked.store.v4f64.p0(<4 x double> undef, ptr align 8 %p, <4 x i1> undef)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 20 for instruction: call void @llvm.masked.store.v8f64.p0(<8 x double> undef, ptr align 8 %p, <8 x i1> undef)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 44 for instruction: call void @llvm.masked.store.v16f64.p0(<16 x double> undef, ptr align 8 %p, <16 x i1> undef)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 92 for instruction: call void @llvm.masked.store.v32f64.p0(<32 x double> undef, ptr align 8 %p, <32 x i1> undef)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; AVX512-LABEL: 'masked_store_f64'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: call void @llvm.masked.store.v4f64.p0(<4 x double> undef, ptr align 8 %p, <4 x i1> undef)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: call void @llvm.masked.store.v8f64.p0(<8 x double> undef, ptr align 8 %p, <8 x i1> undef)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 6 for instruction: call void @llvm.masked.store.v16f64.p0(<16 x double> undef, ptr align 8 %p, <16 x i1> undef)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 12 for instruction: call void @llvm.masked.store.v32f64.p0(<32 x double> undef, ptr align 8 %p, <32 x i1> undef)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+ call void @llvm.masked.store.v4f64.p0(<4 x double> undef, ptr %p, i32 8, <4 x i1> undef)
+ call void @llvm.masked.store.v8f64.p0(<8 x double> undef, ptr %p, i32 8, <8 x i1> undef)
+ call void @llvm.masked.store.v16f64.p0(<16 x double> undef, ptr %p, i32 8, <16 x i1> undef)
+ call void @llvm.masked.store.v32f64.p0(<32 x double> undef, ptr %p, i32 8, <32 x i1> undef)
+ ret void
+}
+
+define void @masked_load_f64(ptr %p) {
+; SSE2-LABEL: 'masked_load_f64'
+; SSE2-NEXT: Cost Model: Found an estimated cost of 17 for instruction: %l4 = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 %p, <4 x i1> undef, <4 x double> undef)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 35 for instruction: %l8 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 %p, <8 x i1> undef, <8 x double> undef)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 71 for instruction: %l16 = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 %p, <16 x i1> undef, <16 x double> undef)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; AVX2-LABEL: 'masked_load_f64'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 2 for instruction: %l4 = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 %p, <4 x i1> undef, <4 x double> undef)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %l8 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 %p, <8 x i1> undef, <8 x double> undef)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 20 for instruction: %l16 = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 %p, <16 x i1> undef, <16 x double> undef)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; AVX512-LABEL: 'masked_load_f64'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %l4 = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 %p, <4 x i1> undef, <4 x double> undef)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %l8 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 %p, <8 x i1> undef, <8 x double> undef)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %l16 = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 %p, <16 x i1> undef, <16 x double> undef)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+ %l4 = call <4 x double> @llvm.masked.load.v4f64.p0(ptr %p, i32 8, <4 x i1> undef, <4 x double> undef)
+ %l8 = call <8 x double> @llvm.masked.load.v8f64.p0(ptr %p, i32 8, <8 x i1> undef, <8 x double> undef)
+ %l16 = call <16 x double> @llvm.masked.load.v16f64.p0(ptr %p, i32 8, <16 x i1> undef, <16 x double> undef)
+ ret void
+}
+
+define void @masked_store_i64(ptr %p) {
+; SSE2-LABEL: 'masked_store_i64'
+; SSE2-NEXT: Cost Model: Found an estimated cost of 21 for instruction: call void @llvm.masked.store.v4i64.p0(<4 x i64> undef, ptr align 8 %p, <4 x i1> undef)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 43 for instruction: call void @llvm.masked.store.v8i64.p0(<8 x i64> undef, ptr align 8 %p, <8 x i1> undef)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 87 for instruction: call void @llvm.masked.store.v16i64.p0(<16 x i64> undef, ptr align 8 %p, <16 x i1> undef)
+; SSE2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; AVX2-LABEL: 'masked_store_i64'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 8 for instruction: call void @llvm.masked.store.v4i64.p0(<4 x i64> undef, ptr align 8 %p, <4 x i1> undef)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 20 for instruction: call void @llvm.masked.store.v8i64.p0(<8 x i64> undef, ptr align 8 %p, <8 x i1> undef)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 44 for instruction: call void @llvm.masked.store.v16i64.p0(<16 x i64> undef, ptr align 8 %p, <16 x i1> undef)
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+; AVX512-LABEL: 'masked_store_i64'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: call void @llvm.masked.store.v4i64.p0(<4 x i64> undef, ptr align 8 %p, <4 x i1> undef)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: call void @llvm.masked.store.v8i64.p0(<8 x i64> undef, ptr align 8 %p, <8 x i1> undef)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 6 for instruction: call void @llvm.masked.store.v16i64.p0(<16 x i64> undef, ptr align 8 %p, <16 x i1> undef)
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret void
+;
+ call void @llvm.masked.store.v4i64.p0(<4 x i64> undef, ptr %p, i32 8, <4 x i1> undef)
+ call void @llvm.masked.store.v8i64.p0(<8 x i64> undef, ptr %p, i32 8, <8 x i1> undef)
+ call void @llvm.masked.store.v16i64.p0(<16 x i64> undef, ptr %p, i32 8, <16 x i1> undef)
+ ret void
+}
diff --git a/llvm/test/Analysis/CostModel/X86/select-mask-expansion.ll b/llvm/test/Analysis/CostModel/X86/select-mask-expansion.ll
new file mode 100644
index 0000000000000..7ae7454bef70c
--- /dev/null
+++ b/llvm/test/Analysis/CostModel/X86/select-mask-expansion.ll
@@ -0,0 +1,63 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+sse2 | FileCheck %s -check-prefixes=CHECK,SSE
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+avx2 | FileCheck %s -check-prefixes=CHECK,AVX2
+; RUN: opt < %s -passes="print<cost-model>" 2>&1 -disable-output -cost-kind=throughput -mtriple=x86_64-- -mattr=+avx512f | FileCheck %s -check-prefixes=CHECK,AVX512
+
+; A vector select whose value type legalizes into more register parts than its
+; <N x i1> condition needs extra kshiftr instructions on AVX-512 to extract a
+; sub-mask for each value part. That fanout is charged only on AVX-512, where
+; the mask lives in a single k-register; SSE/AVX keep a full-width vector mask
+; and are not charged.
+
+define <8 x i64> @sel_v8i64(<8 x i1> %m, <8 x i64> %a, <8 x i64> %b) {
+; SSE-LABEL: 'sel_v8i64'
+; SSE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %s = select <8 x i1> %m, <8 x i64> %a, <8 x i64> %b
+; SSE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i64> %s
+;
+; AVX2-LABEL: 'sel_v8i64'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %s = select <8 x i1> %m, <8 x i64> %a, <8 x i64> %b
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i64> %s
+;
+; AVX512-LABEL: 'sel_v8i64'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %s = select <8 x i1> %m, <8 x i64> %a, <8 x i64> %b
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <8 x i64> %s
+;
+ %s = select <8 x i1> %m, <8 x i64> %a, <8 x i64> %b
+ ret <8 x i64> %s
+}
+
+define <16 x i64> @sel_v16i64(<16 x i1> %m, <16 x i64> %a, <16 x i64> %b) {
+; SSE-LABEL: 'sel_v16i64'
+; SSE-NEXT: Cost Model: Found an estimated cost of 16 for instruction: %s = select <16 x i1> %m, <16 x i64> %a, <16 x i64> %b
+; SSE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <16 x i64> %s
+;
+; AVX2-LABEL: 'sel_v16i64'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %s = select <16 x i1> %m, <16 x i64> %a, <16 x i64> %b
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <16 x i64> %s
+;
+; AVX512-LABEL: 'sel_v16i64'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 6 for instruction: %s = select <16 x i1> %m, <16 x i64> %a, <16 x i64> %b
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <16 x i64> %s
+;
+ %s = select <16 x i1> %m, <16 x i64> %a, <16 x i64> %b
+ ret <16 x i64> %s
+}
+
+define <16 x i32> @sel_v16i32(<16 x i1> %m, <16 x i32> %a, <16 x i32> %b) {
+; SSE-LABEL: 'sel_v16i32'
+; SSE-NEXT: Cost Model: Found an estimated cost of 8 for instruction: %s = select <16 x i1> %m, <16 x i32> %a, <16 x i32> %b
+; SSE-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <16 x i32> %s
+;
+; AVX2-LABEL: 'sel_v16i32'
+; AVX2-NEXT: Cost Model: Found an estimated cost of 4 for instruction: %s = select <16 x i1> %m, <16 x i32> %a, <16 x i32> %b
+; AVX2-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <16 x i32> %s
+;
+; AVX512-LABEL: 'sel_v16i32'
+; AVX512-NEXT: Cost Model: Found an estimated cost of 1 for instruction: %s = select <16 x i1> %m, <16 x i32> %a, <16 x i32> %b
+; AVX512-NEXT: Cost Model: Found an estimated cost of 0 for instruction: ret <16 x i32> %s
+;
+ %s = select <16 x i1> %m, <16 x i32> %a, <16 x i32> %b
+ ret <16 x i32> %s
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK: {{.*}}
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
index 2d5a30019bacd..12e6d1146ba03 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i32-with-i8-index.ll
@@ -51,8 +51,8 @@ define void @test() {
; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB> = load ir<%inB>
-; AVX2-FASTGATHER: Cost of 24 for VF 16: WIDEN ir<%valB> = load ir<%inB>
-; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB> = load ir<%inB>
+; AVX2-FASTGATHER: Cost of 28 for VF 16: WIDEN ir<%valB> = load ir<%inB>
+; AVX2-FASTGATHER: Cost of 60 for VF 32: WIDEN ir<%valB> = load ir<%inB>
;
; AVX512-LABEL: 'test'
; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i32, ptr %inB, align 4
@@ -60,8 +60,8 @@ define void @test() {
; AVX512: Cost of 13 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB> = load ir<%inB>
; AVX512: Cost of 18 for VF 16: WIDEN ir<%valB> = load ir<%inB>
-; AVX512: Cost of 36 for VF 32: WIDEN ir<%valB> = load ir<%inB>
-; AVX512: Cost of 72 for VF 64: WIDEN ir<%valB> = load ir<%inB>
+; AVX512: Cost of 40 for VF 32: WIDEN ir<%valB> = load ir<%inB>
+; AVX512: Cost of 84 for VF 64: WIDEN ir<%valB> = load ir<%inB>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
index ce5828a46eea7..9e0e7f4ffc777 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/gather-i64-with-i8-index.ll
@@ -50,18 +50,18 @@ define void @test() {
; AVX2-FASTGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i64, ptr %inB, align 8
; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB> = load ir<%inB>
; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB> = load ir<%inB>
-; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB> = load ir<%inB>
-; AVX2-FASTGATHER: Cost of 24 for VF 16: WIDEN ir<%valB> = load ir<%inB>
-; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB> = load ir<%inB>
+; AVX2-FASTGATHER: Cost of 16 for VF 8: WIDEN ir<%valB> = load ir<%inB>
+; AVX2-FASTGATHER: Cost of 36 for VF 16: WIDEN ir<%valB> = load ir<%inB>
+; AVX2-FASTGATHER: Cost of 76 for VF 32: WIDEN ir<%valB> = load ir<%inB>
;
; AVX512-LABEL: 'test'
; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB = load i64, ptr %inB, align 8
; AVX512: Cost of 6 for VF 2: REPLICATE ir<%valB> = load ir<%inB>
; AVX512: Cost of 14 for VF 4: REPLICATE ir<%valB> = load ir<%inB>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB> = load ir<%inB>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%valB> = load ir<%inB>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%valB> = load ir<%inB>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%valB> = load ir<%inB>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%valB> = load ir<%inB>
+; AVX512: Cost of 52 for VF 32: WIDEN ir<%valB> = load ir<%inB>
+; AVX512: Cost of 108 for VF 64: WIDEN ir<%valB> = load ir<%inB>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
index 7d18d05a344a8..c9a15e6920702 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-5.ll
@@ -54,21 +54,21 @@ define void @test() {
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v2> = load ir<%in2>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v3> = load ir<%in3>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v4> = load ir<%in4>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
index 909e0a9c8ed18..cf4322f3580a0 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-6.ll
@@ -75,24 +75,24 @@ define void @test() {
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v3> = load ir<%in3>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v4> = load ir<%in4>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v5> = load ir<%in5>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
index a9cef6bd260a1..54a4260a98423 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-7.ll
@@ -59,27 +59,27 @@ define void @test() {
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v4> = load ir<%in4>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v5> = load ir<%in5>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v6> = load ir<%in6>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v6> = load ir<%in6>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v6> = load ir<%in6>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v6> = load ir<%in6>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v6> = load ir<%in6>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v6> = load ir<%in6>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v6> = load ir<%in6>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
index ba511ad467b1a..a254aa8646c03 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-f64-stride-8.ll
@@ -62,30 +62,30 @@ define void @test() {
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v5> = load ir<%in5>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v6> = load ir<%in6>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v7> = load ir<%in7>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v6> = load ir<%in6>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v7> = load ir<%in7>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v6> = load ir<%in6>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v7> = load ir<%in7>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v6> = load ir<%in6>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v7> = load ir<%in7>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v6> = load ir<%in6>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v7> = load ir<%in7>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v6> = load ir<%in6>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v7> = load ir<%in7>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v6> = load ir<%in6>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v7> = load ir<%in7>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
index d0a99efab706d..3a56326dd79da 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-2.ll
@@ -63,10 +63,10 @@ define void @test() {
; AVX512: Cost of 34 for VF 16: INTERLEAVE-GROUP with factor 2, ir<%in0>
; AVX512: ir<%v0> = load from index 0
; AVX512: ir<%v1> = load from index 1
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v1> = load ir<%in1>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
index 180ef142675f5..7fdb62bdd5608 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-4.ll
@@ -68,18 +68,18 @@ define void @test() {
; AVX512: ir<%v1> = load from index 1
; AVX512: ir<%v2> = load from index 2
; AVX512: ir<%v3> = load from index 3
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v3> = load ir<%in3>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
index 65a27fc96f223..11a08c67218e7 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-5.ll
@@ -54,21 +54,21 @@ define void @test() {
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v2> = load ir<%in2>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v3> = load ir<%in3>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v4> = load ir<%in4>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
index d261e5242f8d8..97f3b0f12226f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-6.ll
@@ -75,24 +75,24 @@ define void @test() {
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v3> = load ir<%in3>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v4> = load ir<%in4>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v5> = load ir<%in5>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
index 04c8db7a58357..1dd2e2aabcbe0 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-7.ll
@@ -59,27 +59,27 @@ define void @test() {
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v4> = load ir<%in4>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v5> = load ir<%in5>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v6> = load ir<%in6>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v6> = load ir<%in6>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v6> = load ir<%in6>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v6> = load ir<%in6>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v6> = load ir<%in6>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v6> = load ir<%in6>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v6> = load ir<%in6>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
index a2b5d1c4df2f8..94214fb00d94c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-load-i64-stride-8.ll
@@ -62,30 +62,30 @@ define void @test() {
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v5> = load ir<%in5>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v6> = load ir<%in6>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%v7> = load ir<%in7>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v6> = load ir<%in6>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%v7> = load ir<%in7>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v6> = load ir<%in6>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%v7> = load ir<%in7>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v0> = load ir<%in0>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v1> = load ir<%in1>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v2> = load ir<%in2>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v3> = load ir<%in3>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v4> = load ir<%in4>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v5> = load ir<%in5>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v6> = load ir<%in6>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%v7> = load ir<%in7>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v6> = load ir<%in6>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%v7> = load ir<%in7>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v6> = load ir<%in6>
+; AVX512: Cost of 48 for VF 32: WIDEN ir<%v7> = load ir<%in7>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v0> = load ir<%in0>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v1> = load ir<%in1>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v2> = load ir<%in2>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v3> = load ir<%in3>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v4> = load ir<%in4>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v5> = load ir<%in5>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v6> = load ir<%in6>
+; AVX512: Cost of 96 for VF 64: WIDEN ir<%v7> = load ir<%in7>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-store-f64-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-store-f64-stride-8.ll
index 22d395647482d..e05d6d8e0843c 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-store-f64-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-store-f64-stride-8.ll
@@ -32,30 +32,30 @@ define void @test() {
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out5>, ir<%v5>
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out6>, ir<%v6>
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out7>, ir<%v7>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out0>, ir<%v0>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out1>, ir<%v1>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out2>, ir<%v2>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out3>, ir<%v3>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out4>, ir<%v4>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out5>, ir<%v5>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out6>, ir<%v6>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out7>, ir<%v7>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out0>, ir<%v0>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out1>, ir<%v1>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out2>, ir<%v2>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out3>, ir<%v3>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out4>, ir<%v4>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out5>, ir<%v5>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out6>, ir<%v6>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out7>, ir<%v7>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out0>, ir<%v0>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out1>, ir<%v1>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out2>, ir<%v2>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out3>, ir<%v3>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out4>, ir<%v4>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out5>, ir<%v5>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out6>, ir<%v6>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out7>, ir<%v7>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out0>, ir<%v0>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out1>, ir<%v1>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out2>, ir<%v2>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out3>, ir<%v3>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out4>, ir<%v4>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out5>, ir<%v5>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out6>, ir<%v6>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out7>, ir<%v7>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out0>, ir<%v0>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out1>, ir<%v1>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out2>, ir<%v2>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out3>, ir<%v3>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out4>, ir<%v4>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out5>, ir<%v5>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out6>, ir<%v6>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out7>, ir<%v7>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out0>, ir<%v0>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out1>, ir<%v1>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out2>, ir<%v2>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out3>, ir<%v3>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out4>, ir<%v4>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out5>, ir<%v5>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out6>, ir<%v6>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out7>, ir<%v7>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-store-i64-stride-8.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-store-i64-stride-8.ll
index 6e808e99d351f..c98478bc90660 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-store-i64-stride-8.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/interleaved-store-i64-stride-8.ll
@@ -170,30 +170,30 @@ define void @test() {
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out5>, ir<%v5>
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out6>, ir<%v6>
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out7>, ir<%v7>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out0>, ir<%v>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out1>, ir<%v1>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out2>, ir<%v2>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out3>, ir<%v3>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out4>, ir<%v4>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out5>, ir<%v5>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out6>, ir<%v6>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out7>, ir<%v7>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out0>, ir<%v>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out1>, ir<%v1>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out2>, ir<%v2>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out3>, ir<%v3>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out4>, ir<%v4>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out5>, ir<%v5>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out6>, ir<%v6>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out7>, ir<%v7>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out0>, ir<%v>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out1>, ir<%v1>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out2>, ir<%v2>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out3>, ir<%v3>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out4>, ir<%v4>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out5>, ir<%v5>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out6>, ir<%v6>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out7>, ir<%v7>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out0>, ir<%v>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out1>, ir<%v1>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out2>, ir<%v2>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out3>, ir<%v3>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out4>, ir<%v4>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out5>, ir<%v5>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out6>, ir<%v6>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out7>, ir<%v7>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out0>, ir<%v>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out1>, ir<%v1>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out2>, ir<%v2>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out3>, ir<%v3>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out4>, ir<%v4>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out5>, ir<%v5>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out6>, ir<%v6>
+; AVX512: Cost of 48 for VF 32: WIDEN store ir<%out7>, ir<%v7>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out0>, ir<%v>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out1>, ir<%v1>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out2>, ir<%v2>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out3>, ir<%v3>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out4>, ir<%v4>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out5>, ir<%v5>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out6>, ir<%v6>
+; AVX512: Cost of 96 for VF 64: WIDEN store ir<%out7>, ir<%v7>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
index 8f5da77027970..f4bfc6530c491 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i32-with-i8-index.ll
@@ -44,8 +44,8 @@ define void @test() {
; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 28 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 60 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
;
; AVX512-LABEL: 'test'
; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
@@ -53,8 +53,8 @@ define void @test() {
; AVX512: Cost of 17 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
; AVX512: Cost of 18 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 36 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 72 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 40 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 84 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
index 6801a549e19c4..d93596a160a60 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-gather-i64-with-i8-index.ll
@@ -43,18 +43,18 @@ define void @test() {
; AVX2-FASTGATHER: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
; AVX2-FASTGATHER: Cost of 4 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
; AVX2-FASTGATHER: Cost of 6 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 12 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX2-FASTGATHER: Cost of 48 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 16 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 36 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX2-FASTGATHER: Cost of 76 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
;
; AVX512-LABEL: 'test'
; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
; AVX512: Cost of 8 for VF 2: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
; AVX512: Cost of 18 for VF 4: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
; AVX512: Cost of 10 for VF 8: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 20 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 40 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
-; AVX512: Cost of 80 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 24 for VF 16: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 52 for VF 32: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
+; AVX512: Cost of 108 for VF 64: WIDEN ir<%valB.loaded> = load ir<%inB>, ir<%canLoad>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
index 9cb0e482da47c..cbd611958bdea 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i16.ll
@@ -45,7 +45,7 @@ define void @test(ptr %B) {
; AVX512: Cost of 1 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX512: Cost of 2 for VF 64: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX512: Cost of 6 for VF 64: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
index fc42ce6e6f73f..4a726609447ce 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i32.ll
@@ -27,16 +27,16 @@ define void @test(ptr %B) {
; AVX1: Cost of 3 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX1: Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX1: Cost of 2 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX1: Cost of 4 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX1: Cost of 8 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX1: Cost of 8 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX1: Cost of 20 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
; AVX2-LABEL: 'test'
; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
; AVX2: Cost of 3 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX2: Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX2: Cost of 2 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX2: Cost of 4 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX2: Cost of 8 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX2: Cost of 8 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX2: Cost of 20 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
; AVX512-LABEL: 'test'
; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i32, ptr %inB, align 4
@@ -44,8 +44,8 @@ define void @test(ptr %B) {
; AVX512: Cost of 1 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX512: Cost of 2 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX512: Cost of 4 for VF 64: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX512: Cost of 6 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX512: Cost of 16 for VF 64: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
index 48c9b01beb888..ee0b1441a836b 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-load-i64.ll
@@ -26,26 +26,26 @@ define void @test(ptr %B) {
; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
; AVX1: Cost of 2 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX1: Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX1: Cost of 4 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX1: Cost of 8 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX1: Cost of 16 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX1: Cost of 8 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX1: Cost of 20 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX1: Cost of 44 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
; AVX2-LABEL: 'test'
; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
; AVX2: Cost of 2 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX2: Cost of 2 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX2: Cost of 4 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX2: Cost of 8 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX2: Cost of 16 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX2: Cost of 8 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX2: Cost of 20 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX2: Cost of 44 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
; AVX512-LABEL: 'test'
; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: %valB.loaded = load i64, ptr %inB, align 8
; AVX512: Cost of 1 for VF 2: WIDEN ir<%valB.loaded> = load vp<[[VP6:%[0-9]+]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 4: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
; AVX512: Cost of 1 for VF 8: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX512: Cost of 2 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX512: Cost of 4 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
-; AVX512: Cost of 8 for VF 64: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX512: Cost of 6 for VF 16: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX512: Cost of 16 for VF 32: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
+; AVX512: Cost of 36 for VF 64: WIDEN ir<%valB.loaded> = load vp<[[VP6]]>, ir<%canLoad>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
index 6959fea2d512b..b529918e2298f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i32-with-i8-index.ll
@@ -52,8 +52,8 @@ define void @test() {
; AVX512: Cost of 10.5 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 18 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 36 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 72 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 84 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
index 41ae89933204e..10305e8a96791 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-scatter-i64-with-i8-index.ll
@@ -51,9 +51,9 @@ define void @test() {
; AVX512: Cost of 5 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 11 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 52 for VF 32: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 108 for VF 64: WIDEN store ir<%out>, ir<%valB>, ir<%canStore>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
index 61436a61dba50..2b4b465b8a07f 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i16.ll
@@ -45,7 +45,7 @@ define void @test(ptr %C) {
; AVX512: Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 2 for VF 64: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 6 for VF 64: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
index 0afea8d1664d5..c04214fc714a8 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i32.ll
@@ -34,16 +34,16 @@ define void @test(ptr %C) {
; AVX1: Cost of 9 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX1: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX1: Cost of 8 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX1: Cost of 16 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX1: Cost of 32 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX1: Cost of 20 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX1: Cost of 44 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX2-LABEL: 'test'
; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
; AVX2: Cost of 9 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX2: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX2: Cost of 8 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX2: Cost of 16 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX2: Cost of 32 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX2: Cost of 20 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX2: Cost of 44 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX512-LABEL: 'test'
; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: store i32 %valB, ptr %out, align 4
@@ -51,8 +51,8 @@ define void @test(ptr %C) {
; AVX512: Cost of 1 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 2 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 4 for VF 64: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 6 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 16 for VF 64: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
index ce2d69fca6a3b..bcda6b507f533 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/masked-store-i64.ll
@@ -33,26 +33,26 @@ define void @test(ptr %C) {
; AVX1: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX1: Cost of 8 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX1: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX1: Cost of 16 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX1: Cost of 32 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX1: Cost of 64 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX1: Cost of 20 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX1: Cost of 44 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX1: Cost of 92 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX2-LABEL: 'test'
; AVX2: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX2: Cost of 8 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX2: Cost of 8 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX2: Cost of 16 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX2: Cost of 32 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX2: Cost of 64 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX2: Cost of 20 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX2: Cost of 44 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX2: Cost of 92 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
; AVX512-LABEL: 'test'
; AVX512: LV: Found an estimated cost of 1 for VF 1 For instruction: store i64 %valB, ptr %out, align 8
; AVX512: Cost of 1 for VF 2: WIDEN store vp<[[VP7:%[0-9]+]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 4: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
; AVX512: Cost of 1 for VF 8: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 2 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 4 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
-; AVX512: Cost of 8 for VF 64: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 6 for VF 16: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 16 for VF 32: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
+; AVX512: Cost of 36 for VF 64: WIDEN store vp<[[VP7]]>, ir<%valB>, ir<%canStore>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
index 5e0b3277dd5e9..078045d450dce 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i32-with-i8-index.ll
@@ -52,8 +52,8 @@ define void @test() {
; AVX512: Cost of 13 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>
; AVX512: Cost of 18 for VF 16: WIDEN store ir<%out>, ir<%valB>
-; AVX512: Cost of 36 for VF 32: WIDEN store ir<%out>, ir<%valB>
-; AVX512: Cost of 72 for VF 64: WIDEN store ir<%out>, ir<%valB>
+; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out>, ir<%valB>
+; AVX512: Cost of 84 for VF 64: WIDEN store ir<%out>, ir<%valB>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
index fcbf6042dec14..cbacfa9566008 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/scatter-i64-with-i8-index.ll
@@ -51,9 +51,9 @@ define void @test() {
; AVX512: Cost of 6 for VF 2: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 14 for VF 4: REPLICATE store ir<%valB>, ir<%out>
; AVX512: Cost of 10 for VF 8: WIDEN store ir<%out>, ir<%valB>
-; AVX512: Cost of 20 for VF 16: WIDEN store ir<%out>, ir<%valB>
-; AVX512: Cost of 40 for VF 32: WIDEN store ir<%out>, ir<%valB>
-; AVX512: Cost of 80 for VF 64: WIDEN store ir<%out>, ir<%valB>
+; AVX512: Cost of 24 for VF 16: WIDEN store ir<%out>, ir<%valB>
+; AVX512: Cost of 52 for VF 32: WIDEN store ir<%out>, ir<%valB>
+; AVX512: Cost of 108 for VF 64: WIDEN store ir<%out>, ir<%valB>
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll b/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll
index 806a81f721212..4ee17d5a922ef 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/masked_load_store.ll
@@ -813,15 +813,42 @@ define void @foo3(ptr nocapture %A, ptr nocapture readonly %B, ptr nocapture rea
; AVX2: [[VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
-; AVX2-NEXT: [[TMP1:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD]], splat (i32 100)
+; AVX2-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 4
+; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 8
+; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 12
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
+; AVX2-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4, !alias.scope [[META12]]
+; AVX2-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META12]]
+; AVX2-NEXT: [[WIDE_LOAD8:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META12]]
+; AVX2-NEXT: [[TMP4:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD]], splat (i32 100)
+; AVX2-NEXT: [[TMP5:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD6]], splat (i32 100)
+; AVX2-NEXT: [[TMP6:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD7]], splat (i32 100)
+; AVX2-NEXT: [[TMP7:%.*]] = icmp slt <4 x i32> [[WIDE_LOAD8]], splat (i32 100)
; AVX2-NEXT: [[TMP8:%.*]] = getelementptr double, ptr [[B]], i64 [[INDEX]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP8]], <8 x i1> [[TMP1]], <8 x double> poison), !alias.scope [[META15:![0-9]+]]
-; AVX2-NEXT: [[TMP3:%.*]] = sitofp <8 x i32> [[WIDE_LOAD]] to <8 x double>
-; AVX2-NEXT: [[TMP4:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], [[TMP3]]
+; AVX2-NEXT: [[TMP9:%.*]] = getelementptr double, ptr [[TMP8]], i64 4
+; AVX2-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[TMP8]], i64 8
+; AVX2-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP8]], i64 12
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP8]], <4 x i1> [[TMP4]], <4 x double> poison), !alias.scope [[META15:![0-9]+]]
+; AVX2-NEXT: [[WIDE_MASKED_LOAD9:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP9]], <4 x i1> [[TMP5]], <4 x double> poison), !alias.scope [[META15]]
+; AVX2-NEXT: [[WIDE_MASKED_LOAD10:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP10]], <4 x i1> [[TMP6]], <4 x double> poison), !alias.scope [[META15]]
+; AVX2-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP11]], <4 x i1> [[TMP7]], <4 x double> poison), !alias.scope [[META15]]
+; AVX2-NEXT: [[TMP12:%.*]] = sitofp <4 x i32> [[WIDE_LOAD]] to <4 x double>
+; AVX2-NEXT: [[TMP13:%.*]] = sitofp <4 x i32> [[WIDE_LOAD6]] to <4 x double>
+; AVX2-NEXT: [[TMP14:%.*]] = sitofp <4 x i32> [[WIDE_LOAD7]] to <4 x double>
+; AVX2-NEXT: [[TMP15:%.*]] = sitofp <4 x i32> [[WIDE_LOAD8]] to <4 x double>
+; AVX2-NEXT: [[TMP16:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD]], [[TMP12]]
+; AVX2-NEXT: [[TMP17:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD9]], [[TMP13]]
+; AVX2-NEXT: [[TMP18:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD10]], [[TMP14]]
+; AVX2-NEXT: [[TMP19:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD11]], [[TMP15]]
; AVX2-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[A]], i64 [[INDEX]]
-; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP4]], ptr align 8 [[TMP20]], <8 x i1> [[TMP1]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; AVX2-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP20]], i64 4
+; AVX2-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP20]], i64 8
+; AVX2-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP20]], i64 12
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP16]], ptr align 8 [[TMP20]], <4 x i1> [[TMP4]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP17]], ptr align 8 [[TMP21]], <4 x i1> [[TMP5]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP18]], ptr align 8 [[TMP22]], <4 x i1> [[TMP6]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP19]], ptr align 8 [[TMP23]], <4 x i1> [[TMP7]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX2-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 10000
; AVX2-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
@@ -851,65 +878,65 @@ define void @foo3(ptr nocapture %A, ptr nocapture readonly %B, ptr nocapture rea
; AVX512: [[VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[INDEX]]
+; AVX512-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 8
; AVX512-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 16
-; AVX512-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 32
-; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 48
-; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
-; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <16 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <16 x i32>, ptr [[TMP9]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <16 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[TMP4:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD]], splat (i32 100)
-; AVX512-NEXT: [[TMP5:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD6]], splat (i32 100)
-; AVX512-NEXT: [[TMP6:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD7]], splat (i32 100)
-; AVX512-NEXT: [[TMP7:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD8]], splat (i32 100)
+; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 24
+; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP0]], align 4, !alias.scope [[META12:![0-9]+]]
+; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <8 x i32>, ptr [[TMP1]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <8 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[TMP4:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD]], splat (i32 100)
+; AVX512-NEXT: [[TMP5:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD6]], splat (i32 100)
+; AVX512-NEXT: [[TMP6:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD7]], splat (i32 100)
+; AVX512-NEXT: [[TMP7:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD8]], splat (i32 100)
; AVX512-NEXT: [[TMP8:%.*]] = getelementptr double, ptr [[B]], i64 [[INDEX]]
+; AVX512-NEXT: [[TMP9:%.*]] = getelementptr double, ptr [[TMP8]], i64 8
; AVX512-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[TMP8]], i64 16
-; AVX512-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP8]], i64 32
-; AVX512-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP8]], i64 48
-; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP8]], <16 x i1> [[TMP4]], <16 x double> poison), !alias.scope [[META15:![0-9]+]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD9:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP10]], <16 x i1> [[TMP5]], <16 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD10:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP21]], <16 x i1> [[TMP6]], <16 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP11]], <16 x i1> [[TMP7]], <16 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[TMP12:%.*]] = sitofp <16 x i32> [[WIDE_LOAD]] to <16 x double>
-; AVX512-NEXT: [[TMP13:%.*]] = sitofp <16 x i32> [[WIDE_LOAD6]] to <16 x double>
-; AVX512-NEXT: [[TMP14:%.*]] = sitofp <16 x i32> [[WIDE_LOAD7]] to <16 x double>
-; AVX512-NEXT: [[TMP15:%.*]] = sitofp <16 x i32> [[WIDE_LOAD8]] to <16 x double>
-; AVX512-NEXT: [[TMP16:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD]], [[TMP12]]
-; AVX512-NEXT: [[TMP17:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD9]], [[TMP13]]
-; AVX512-NEXT: [[TMP18:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD10]], [[TMP14]]
-; AVX512-NEXT: [[TMP19:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD11]], [[TMP15]]
+; AVX512-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP8]], i64 24
+; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP8]], <8 x i1> [[TMP4]], <8 x double> poison), !alias.scope [[META15:![0-9]+]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD9:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP9]], <8 x i1> [[TMP5]], <8 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD10:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP10]], <8 x i1> [[TMP6]], <8 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP11]], <8 x i1> [[TMP7]], <8 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[TMP12:%.*]] = sitofp <8 x i32> [[WIDE_LOAD]] to <8 x double>
+; AVX512-NEXT: [[TMP13:%.*]] = sitofp <8 x i32> [[WIDE_LOAD6]] to <8 x double>
+; AVX512-NEXT: [[TMP14:%.*]] = sitofp <8 x i32> [[WIDE_LOAD7]] to <8 x double>
+; AVX512-NEXT: [[TMP15:%.*]] = sitofp <8 x i32> [[WIDE_LOAD8]] to <8 x double>
+; AVX512-NEXT: [[TMP16:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], [[TMP12]]
+; AVX512-NEXT: [[TMP17:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD9]], [[TMP13]]
+; AVX512-NEXT: [[TMP18:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD10]], [[TMP14]]
+; AVX512-NEXT: [[TMP19:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD11]], [[TMP15]]
; AVX512-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[A]], i64 [[INDEX]]
+; AVX512-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP20]], i64 8
; AVX512-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP20]], i64 16
-; AVX512-NEXT: [[TMP32:%.*]] = getelementptr double, ptr [[TMP20]], i64 32
-; AVX512-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP20]], i64 48
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP16]], ptr align 8 [[TMP20]], <16 x i1> [[TMP4]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP17]], ptr align 8 [[TMP22]], <16 x i1> [[TMP5]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP18]], ptr align 8 [[TMP32]], <16 x i1> [[TMP6]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP19]], ptr align 8 [[TMP23]], <16 x i1> [[TMP7]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
+; AVX512-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP20]], i64 24
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP16]], ptr align 8 [[TMP20]], <8 x i1> [[TMP4]]), !alias.scope [[META17:![0-9]+]], !noalias [[META19:![0-9]+]]
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP17]], ptr align 8 [[TMP21]], <8 x i1> [[TMP5]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP18]], ptr align 8 [[TMP22]], <8 x i1> [[TMP6]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP19]], ptr align 8 [[TMP23]], <8 x i1> [[TMP7]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; AVX512-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 9984
; AVX512-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: br i1 false, [[FOR_END:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX512: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX512-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
+; AVX512-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF21:![0-9]+]]
; AVX512: [[VEC_EPILOG_PH]]:
; AVX512-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ 9984, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX512-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX512: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX12:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT15:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP25:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[INDEX12]]
-; AVX512-NEXT: [[WIDE_LOAD13:%.*]] = load <16 x i32>, ptr [[TMP25]], align 4, !alias.scope [[META12]]
-; AVX512-NEXT: [[TMP26:%.*]] = icmp slt <16 x i32> [[WIDE_LOAD13]], splat (i32 100)
+; AVX512-NEXT: [[WIDE_LOAD13:%.*]] = load <8 x i32>, ptr [[TMP25]], align 4, !alias.scope [[META12]]
+; AVX512-NEXT: [[TMP26:%.*]] = icmp slt <8 x i32> [[WIDE_LOAD13]], splat (i32 100)
; AVX512-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[B]], i64 [[INDEX12]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD14:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP27]], <16 x i1> [[TMP26]], <16 x double> poison), !alias.scope [[META15]]
-; AVX512-NEXT: [[TMP28:%.*]] = sitofp <16 x i32> [[WIDE_LOAD13]] to <16 x double>
-; AVX512-NEXT: [[TMP29:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD14]], [[TMP28]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD14:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP27]], <8 x i1> [[TMP26]], <8 x double> poison), !alias.scope [[META15]]
+; AVX512-NEXT: [[TMP28:%.*]] = sitofp <8 x i32> [[WIDE_LOAD13]] to <8 x double>
+; AVX512-NEXT: [[TMP29:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD14]], [[TMP28]]
; AVX512-NEXT: [[TMP30:%.*]] = getelementptr double, ptr [[A]], i64 [[INDEX12]]
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP29]], ptr align 8 [[TMP30]], <16 x i1> [[TMP26]]), !alias.scope [[META17]], !noalias [[META19]]
-; AVX512-NEXT: [[INDEX_NEXT15]] = add nuw i64 [[INDEX12]], 16
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP29]], ptr align 8 [[TMP30]], <8 x i1> [[TMP26]]), !alias.scope [[META17]], !noalias [[META19]]
+; AVX512-NEXT: [[INDEX_NEXT15]] = add nuw i64 [[INDEX12]], 8
; AVX512-NEXT: [[TMP31:%.*]] = icmp eq i64 [[INDEX_NEXT15]], 10000
-; AVX512-NEXT: br i1 [[TMP31]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP21:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP31]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP22:![0-9]+]]
; AVX512: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX512-NEXT: br i1 true, [[FOR_END]], label %[[VEC_EPILOG_SCALAR_PH]]
; AVX512: [[VEC_EPILOG_SCALAR_PH]]:
@@ -1000,21 +1027,21 @@ define void @foo4(ptr nocapture %A, ptr nocapture readonly %B, ptr nocapture rea
; AVX512-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX512: [[VECTOR_BODY]]:
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; AVX512-NEXT: [[VEC_IND:%.*]] = phi <16 x i64> [ <i64 0, i64 16, i64 32, i64 48, i64 64, i64 80, i64 96, i64 112, i64 128, i64 144, i64 160, i64 176, i64 192, i64 208, i64 224, i64 240>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; AVX512-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], <16 x i64> [[VEC_IND]]
-; AVX512-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <16 x i32> @llvm.masked.gather.v16i32.v16p0(<16 x ptr> align 4 [[WIDE_GEP]], <16 x i1> splat (i1 true), <16 x i32> poison), !alias.scope [[META23:![0-9]+]]
-; AVX512-NEXT: [[TMP0:%.*]] = icmp slt <16 x i32> [[WIDE_MASKED_GATHER]], splat (i32 100)
-; AVX512-NEXT: [[TMP1:%.*]] = shl nuw nsw <16 x i64> [[VEC_IND]], splat (i64 1)
-; AVX512-NEXT: [[WIDE_GEP6:%.*]] = getelementptr inbounds double, ptr [[B]], <16 x i64> [[TMP1]]
-; AVX512-NEXT: [[WIDE_MASKED_GATHER7:%.*]] = call <16 x double> @llvm.masked.gather.v16f64.v16p0(<16 x ptr> align 8 [[WIDE_GEP6]], <16 x i1> [[TMP0]], <16 x double> poison), !alias.scope [[META26:![0-9]+]]
-; AVX512-NEXT: [[TMP2:%.*]] = sitofp <16 x i32> [[WIDE_MASKED_GATHER]] to <16 x double>
-; AVX512-NEXT: [[TMP3:%.*]] = fadd <16 x double> [[WIDE_MASKED_GATHER7]], [[TMP2]]
-; AVX512-NEXT: [[WIDE_GEP8:%.*]] = getelementptr inbounds double, ptr [[A]], <16 x i64> [[VEC_IND]]
-; AVX512-NEXT: call void @llvm.masked.scatter.v16f64.v16p0(<16 x double> [[TMP3]], <16 x ptr> align 8 [[WIDE_GEP8]], <16 x i1> [[TMP0]]), !alias.scope [[META28:![0-9]+]], !noalias [[META30:![0-9]+]]
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
-; AVX512-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <16 x i64> [[VEC_IND]], splat (i64 256)
+; AVX512-NEXT: [[VEC_IND:%.*]] = phi <8 x i64> [ <i64 0, i64 16, i64 32, i64 48, i64 64, i64 80, i64 96, i64 112>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; AVX512-NEXT: [[WIDE_GEP:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], <8 x i64> [[VEC_IND]]
+; AVX512-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <8 x i32> @llvm.masked.gather.v8i32.v8p0(<8 x ptr> align 4 [[WIDE_GEP]], <8 x i1> splat (i1 true), <8 x i32> poison), !alias.scope [[META24:![0-9]+]]
+; AVX512-NEXT: [[TMP0:%.*]] = icmp slt <8 x i32> [[WIDE_MASKED_GATHER]], splat (i32 100)
+; AVX512-NEXT: [[TMP1:%.*]] = shl nuw nsw <8 x i64> [[VEC_IND]], splat (i64 1)
+; AVX512-NEXT: [[WIDE_GEP6:%.*]] = getelementptr inbounds double, ptr [[B]], <8 x i64> [[TMP1]]
+; AVX512-NEXT: [[WIDE_MASKED_GATHER7:%.*]] = call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 8 [[WIDE_GEP6]], <8 x i1> [[TMP0]], <8 x double> poison), !alias.scope [[META27:![0-9]+]]
+; AVX512-NEXT: [[TMP2:%.*]] = sitofp <8 x i32> [[WIDE_MASKED_GATHER]] to <8 x double>
+; AVX512-NEXT: [[TMP3:%.*]] = fadd <8 x double> [[WIDE_MASKED_GATHER7]], [[TMP2]]
+; AVX512-NEXT: [[WIDE_GEP8:%.*]] = getelementptr inbounds double, ptr [[A]], <8 x i64> [[VEC_IND]]
+; AVX512-NEXT: call void @llvm.masked.scatter.v8f64.v8p0(<8 x double> [[TMP3]], <8 x ptr> align 8 [[WIDE_GEP8]], <8 x i1> [[TMP0]]), !alias.scope [[META29:![0-9]+]], !noalias [[META31:![0-9]+]]
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; AVX512-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <8 x i64> [[VEC_IND]], splat (i64 128)
; AVX512-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], 624
-; AVX512-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP31:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: br label %[[SCALAR_PH]]
; AVX512: [[SCALAR_PH]]:
@@ -1084,19 +1111,49 @@ define void @foo6(ptr nocapture readonly %in, ptr nocapture %out, i32 %size, ptr
; AVX1-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP0:%.*]] = sub i64 4095, [[INDEX]]
; AVX1-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[TMP0]]
+; AVX1-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -3
; AVX1-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -7
-; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META18:![0-9]+]]
-; AVX1-NEXT: [[REVERSE:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[TMP4:%.*]] = icmp sgt <8 x i32> [[REVERSE]], zeroinitializer
+; AVX1-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -11
+; AVX1-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -15
+; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META18:![0-9]+]]
+; AVX1-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META18]]
+; AVX1-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META18]]
+; AVX1-NEXT: [[WIDE_LOAD8:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META18]]
+; AVX1-NEXT: [[REVERSE:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[REVERSE9:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD6]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[REVERSE10:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD7]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[REVERSE11:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD8]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[TMP6:%.*]] = icmp sgt <4 x i32> [[REVERSE]], zeroinitializer
+; AVX1-NEXT: [[TMP7:%.*]] = icmp sgt <4 x i32> [[REVERSE9]], zeroinitializer
+; AVX1-NEXT: [[TMP8:%.*]] = icmp sgt <4 x i32> [[REVERSE10]], zeroinitializer
+; AVX1-NEXT: [[TMP9:%.*]] = icmp sgt <4 x i32> [[REVERSE11]], zeroinitializer
; AVX1-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[IN]], i64 [[TMP0]]
+; AVX1-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP10]], i64 -3
; AVX1-NEXT: [[TMP12:%.*]] = getelementptr double, ptr [[TMP10]], i64 -7
-; AVX1-NEXT: [[REVERSE6:%.*]] = shufflevector <8 x i1> [[TMP4]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP12]], <8 x i1> [[REVERSE6]], <8 x double> poison), !alias.scope [[META21:![0-9]+]]
-; AVX1-NEXT: [[TMP6:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
+; AVX1-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -11
+; AVX1-NEXT: [[TMP14:%.*]] = getelementptr double, ptr [[TMP10]], i64 -15
+; AVX1-NEXT: [[REVERSE12:%.*]] = shufflevector <4 x i1> [[TMP6]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[REVERSE13:%.*]] = shufflevector <4 x i1> [[TMP7]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[REVERSE14:%.*]] = shufflevector <4 x i1> [[TMP8]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[REVERSE15:%.*]] = shufflevector <4 x i1> [[TMP9]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP11]], <4 x i1> [[REVERSE12]], <4 x double> poison), !alias.scope [[META21:![0-9]+]]
+; AVX1-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP12]], <4 x i1> [[REVERSE13]], <4 x double> poison), !alias.scope [[META21]]
+; AVX1-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP13]], <4 x i1> [[REVERSE14]], <4 x double> poison), !alias.scope [[META21]]
+; AVX1-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP14]], <4 x i1> [[REVERSE15]], <4 x double> poison), !alias.scope [[META21]]
+; AVX1-NEXT: [[TMP15:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
+; AVX1-NEXT: [[TMP16:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
+; AVX1-NEXT: [[TMP17:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
+; AVX1-NEXT: [[TMP18:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
; AVX1-NEXT: [[TMP19:%.*]] = getelementptr double, ptr [[OUT]], i64 [[TMP0]]
+; AVX1-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP19]], i64 -3
; AVX1-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP19]], i64 -7
-; AVX1-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP6]], ptr align 8 [[TMP21]], <8 x i1> [[REVERSE6]]), !alias.scope [[META23:![0-9]+]], !noalias [[META25:![0-9]+]]
-; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; AVX1-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -11
+; AVX1-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP19]], i64 -15
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP15]], ptr align 8 [[TMP20]], <4 x i1> [[REVERSE12]]), !alias.scope [[META23:![0-9]+]], !noalias [[META25:![0-9]+]]
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP16]], ptr align 8 [[TMP21]], <4 x i1> [[REVERSE13]]), !alias.scope [[META23]], !noalias [[META25]]
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP17]], ptr align 8 [[TMP22]], <4 x i1> [[REVERSE14]]), !alias.scope [[META23]], !noalias [[META25]]
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP18]], ptr align 8 [[TMP23]], <4 x i1> [[REVERSE15]]), !alias.scope [[META23]], !noalias [[META25]]
+; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX1-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4096
; AVX1-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP26:![0-9]+]]
; AVX1: [[MIDDLE_BLOCK]]:
@@ -1125,19 +1182,49 @@ define void @foo6(ptr nocapture readonly %in, ptr nocapture %out, i32 %size, ptr
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = sub i64 4095, [[INDEX]]
; AVX2-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[TMP0]]
+; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -3
; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -7
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META22:![0-9]+]]
-; AVX2-NEXT: [[REVERSE:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[TMP4:%.*]] = icmp sgt <8 x i32> [[REVERSE]], zeroinitializer
+; AVX2-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -11
+; AVX2-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -15
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META22:![0-9]+]]
+; AVX2-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META22]]
+; AVX2-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META22]]
+; AVX2-NEXT: [[WIDE_LOAD8:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META22]]
+; AVX2-NEXT: [[REVERSE:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[REVERSE9:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD6]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[REVERSE10:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD7]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[REVERSE11:%.*]] = shufflevector <4 x i32> [[WIDE_LOAD8]], <4 x i32> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[TMP6:%.*]] = icmp sgt <4 x i32> [[REVERSE]], zeroinitializer
+; AVX2-NEXT: [[TMP7:%.*]] = icmp sgt <4 x i32> [[REVERSE9]], zeroinitializer
+; AVX2-NEXT: [[TMP8:%.*]] = icmp sgt <4 x i32> [[REVERSE10]], zeroinitializer
+; AVX2-NEXT: [[TMP9:%.*]] = icmp sgt <4 x i32> [[REVERSE11]], zeroinitializer
; AVX2-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[IN]], i64 [[TMP0]]
+; AVX2-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP10]], i64 -3
; AVX2-NEXT: [[TMP12:%.*]] = getelementptr double, ptr [[TMP10]], i64 -7
-; AVX2-NEXT: [[REVERSE6:%.*]] = shufflevector <8 x i1> [[TMP4]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP12]], <8 x i1> [[REVERSE6]], <8 x double> poison), !alias.scope [[META25:![0-9]+]]
-; AVX2-NEXT: [[TMP6:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
+; AVX2-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -11
+; AVX2-NEXT: [[TMP14:%.*]] = getelementptr double, ptr [[TMP10]], i64 -15
+; AVX2-NEXT: [[REVERSE12:%.*]] = shufflevector <4 x i1> [[TMP6]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[REVERSE13:%.*]] = shufflevector <4 x i1> [[TMP7]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[REVERSE14:%.*]] = shufflevector <4 x i1> [[TMP8]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[REVERSE15:%.*]] = shufflevector <4 x i1> [[TMP9]], <4 x i1> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP11]], <4 x i1> [[REVERSE12]], <4 x double> poison), !alias.scope [[META25:![0-9]+]]
+; AVX2-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP12]], <4 x i1> [[REVERSE13]], <4 x double> poison), !alias.scope [[META25]]
+; AVX2-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP13]], <4 x i1> [[REVERSE14]], <4 x double> poison), !alias.scope [[META25]]
+; AVX2-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <4 x double> @llvm.masked.load.v4f64.p0(ptr align 8 [[TMP14]], <4 x i1> [[REVERSE15]], <4 x double> poison), !alias.scope [[META25]]
+; AVX2-NEXT: [[TMP15:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
+; AVX2-NEXT: [[TMP16:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
+; AVX2-NEXT: [[TMP17:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
+; AVX2-NEXT: [[TMP18:%.*]] = fadd <4 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
; AVX2-NEXT: [[TMP19:%.*]] = getelementptr double, ptr [[OUT]], i64 [[TMP0]]
+; AVX2-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP19]], i64 -3
; AVX2-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP19]], i64 -7
-; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP6]], ptr align 8 [[TMP21]], <8 x i1> [[REVERSE6]]), !alias.scope [[META27:![0-9]+]], !noalias [[META29:![0-9]+]]
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; AVX2-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -11
+; AVX2-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP19]], i64 -15
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP15]], ptr align 8 [[TMP20]], <4 x i1> [[REVERSE12]]), !alias.scope [[META27:![0-9]+]], !noalias [[META29:![0-9]+]]
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP16]], ptr align 8 [[TMP21]], <4 x i1> [[REVERSE13]]), !alias.scope [[META27]], !noalias [[META29]]
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP17]], ptr align 8 [[TMP22]], <4 x i1> [[REVERSE14]]), !alias.scope [[META27]], !noalias [[META29]]
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> [[TMP18]], ptr align 8 [[TMP23]], <4 x i1> [[REVERSE15]]), !alias.scope [[META27]], !noalias [[META29]]
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX2-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4096
; AVX2-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP30:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
@@ -1166,51 +1253,51 @@ define void @foo6(ptr nocapture readonly %in, ptr nocapture %out, i32 %size, ptr
; AVX512-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX512-NEXT: [[TMP0:%.*]] = sub i64 4095, [[INDEX]]
; AVX512-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TRIGGER]], i64 [[TMP0]]
+; AVX512-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -7
; AVX512-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -15
+; AVX512-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -23
; AVX512-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -31
-; AVX512-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -47
-; AVX512-NEXT: [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 -63
-; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META33:![0-9]+]]
-; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <16 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META33]]
-; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <16 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META33]]
-; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <16 x i32>, ptr [[TMP11]], align 4, !alias.scope [[META33]]
-; AVX512-NEXT: [[REVERSE:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE9:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD6]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE10:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD7]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE11:%.*]] = shufflevector <16 x i32> [[WIDE_LOAD8]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[TMP6:%.*]] = icmp sgt <16 x i32> [[REVERSE]], zeroinitializer
-; AVX512-NEXT: [[TMP7:%.*]] = icmp sgt <16 x i32> [[REVERSE9]], zeroinitializer
-; AVX512-NEXT: [[TMP8:%.*]] = icmp sgt <16 x i32> [[REVERSE10]], zeroinitializer
-; AVX512-NEXT: [[TMP9:%.*]] = icmp sgt <16 x i32> [[REVERSE11]], zeroinitializer
+; AVX512-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP2]], align 4, !alias.scope [[META34:![0-9]+]]
+; AVX512-NEXT: [[WIDE_LOAD6:%.*]] = load <8 x i32>, ptr [[TMP3]], align 4, !alias.scope [[META34]]
+; AVX512-NEXT: [[WIDE_LOAD7:%.*]] = load <8 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META34]]
+; AVX512-NEXT: [[WIDE_LOAD8:%.*]] = load <8 x i32>, ptr [[TMP5]], align 4, !alias.scope [[META34]]
+; AVX512-NEXT: [[REVERSE:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE9:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD6]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE10:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD7]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE11:%.*]] = shufflevector <8 x i32> [[WIDE_LOAD8]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[TMP6:%.*]] = icmp sgt <8 x i32> [[REVERSE]], zeroinitializer
+; AVX512-NEXT: [[TMP7:%.*]] = icmp sgt <8 x i32> [[REVERSE9]], zeroinitializer
+; AVX512-NEXT: [[TMP8:%.*]] = icmp sgt <8 x i32> [[REVERSE10]], zeroinitializer
+; AVX512-NEXT: [[TMP9:%.*]] = icmp sgt <8 x i32> [[REVERSE11]], zeroinitializer
; AVX512-NEXT: [[TMP10:%.*]] = getelementptr double, ptr [[IN]], i64 [[TMP0]]
+; AVX512-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[TMP10]], i64 -7
; AVX512-NEXT: [[TMP12:%.*]] = getelementptr double, ptr [[TMP10]], i64 -15
+; AVX512-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -23
; AVX512-NEXT: [[TMP14:%.*]] = getelementptr double, ptr [[TMP10]], i64 -31
-; AVX512-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[TMP10]], i64 -47
-; AVX512-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP10]], i64 -63
-; AVX512-NEXT: [[REVERSE12:%.*]] = shufflevector <16 x i1> [[TMP6]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE13:%.*]] = shufflevector <16 x i1> [[TMP7]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE14:%.*]] = shufflevector <16 x i1> [[TMP8]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[REVERSE15:%.*]] = shufflevector <16 x i1> [[TMP9]], <16 x i1> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP12]], <16 x i1> [[REVERSE12]], <16 x double> poison), !alias.scope [[META36:![0-9]+]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP14]], <16 x i1> [[REVERSE13]], <16 x double> poison), !alias.scope [[META36]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP13]], <16 x i1> [[REVERSE14]], <16 x double> poison), !alias.scope [[META36]]
-; AVX512-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <16 x double> @llvm.masked.load.v16f64.p0(ptr align 8 [[TMP20]], <16 x i1> [[REVERSE15]], <16 x double> poison), !alias.scope [[META36]]
-; AVX512-NEXT: [[TMP15:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
-; AVX512-NEXT: [[TMP16:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
-; AVX512-NEXT: [[TMP17:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
-; AVX512-NEXT: [[TMP18:%.*]] = fadd <16 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[REVERSE12:%.*]] = shufflevector <8 x i1> [[TMP6]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE13:%.*]] = shufflevector <8 x i1> [[TMP7]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE14:%.*]] = shufflevector <8 x i1> [[TMP8]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[REVERSE15:%.*]] = shufflevector <8 x i1> [[TMP9]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; AVX512-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP11]], <8 x i1> [[REVERSE12]], <8 x double> poison), !alias.scope [[META37:![0-9]+]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD16:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP12]], <8 x i1> [[REVERSE13]], <8 x double> poison), !alias.scope [[META37]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD17:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP13]], <8 x i1> [[REVERSE14]], <8 x double> poison), !alias.scope [[META37]]
+; AVX512-NEXT: [[WIDE_MASKED_LOAD18:%.*]] = call <8 x double> @llvm.masked.load.v8f64.p0(ptr align 8 [[TMP14]], <8 x i1> [[REVERSE15]], <8 x double> poison), !alias.scope [[META37]]
+; AVX512-NEXT: [[TMP15:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[TMP16:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD16]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[TMP17:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD17]], splat (double 5.000000e-01)
+; AVX512-NEXT: [[TMP18:%.*]] = fadd <8 x double> [[WIDE_MASKED_LOAD18]], splat (double 5.000000e-01)
; AVX512-NEXT: [[TMP19:%.*]] = getelementptr double, ptr [[OUT]], i64 [[TMP0]]
+; AVX512-NEXT: [[TMP20:%.*]] = getelementptr double, ptr [[TMP19]], i64 -7
; AVX512-NEXT: [[TMP21:%.*]] = getelementptr double, ptr [[TMP19]], i64 -15
+; AVX512-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -23
; AVX512-NEXT: [[TMP23:%.*]] = getelementptr double, ptr [[TMP19]], i64 -31
-; AVX512-NEXT: [[TMP22:%.*]] = getelementptr double, ptr [[TMP19]], i64 -47
-; AVX512-NEXT: [[TMP25:%.*]] = getelementptr double, ptr [[TMP19]], i64 -63
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP15]], ptr align 8 [[TMP21]], <16 x i1> [[REVERSE12]]), !alias.scope [[META38:![0-9]+]], !noalias [[META40:![0-9]+]]
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP16]], ptr align 8 [[TMP23]], <16 x i1> [[REVERSE13]]), !alias.scope [[META38]], !noalias [[META40]]
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP17]], ptr align 8 [[TMP22]], <16 x i1> [[REVERSE14]]), !alias.scope [[META38]], !noalias [[META40]]
-; AVX512-NEXT: call void @llvm.masked.store.v16f64.p0(<16 x double> [[TMP18]], ptr align 8 [[TMP25]], <16 x i1> [[REVERSE15]]), !alias.scope [[META38]], !noalias [[META40]]
-; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP15]], ptr align 8 [[TMP20]], <8 x i1> [[REVERSE12]]), !alias.scope [[META39:![0-9]+]], !noalias [[META41:![0-9]+]]
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP16]], ptr align 8 [[TMP21]], <8 x i1> [[REVERSE13]]), !alias.scope [[META39]], !noalias [[META41]]
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP17]], ptr align 8 [[TMP22]], <8 x i1> [[REVERSE14]]), !alias.scope [[META39]], !noalias [[META41]]
+; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> [[TMP18]], ptr align 8 [[TMP23]], <8 x i1> [[REVERSE15]]), !alias.scope [[META39]], !noalias [[META41]]
+; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
; AVX512-NEXT: [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4096
-; AVX512-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP41:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP42:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: br [[FOR_END:label %.*]]
; AVX512: [[SCALAR_PH]]:
@@ -1258,54 +1345,84 @@ define void @foo7(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX1-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX1: [[ITER_CHECK]]:
; AVX1-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
+; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX1: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
+; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX1: [[VECTOR_PH]]:
-; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
+; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
; AVX1-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX1-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX1: [[VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
-; AVX1-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX1-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
+; AVX1-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
+; AVX1-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
+; AVX1-NEXT: [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
+; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
+; AVX1-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
+; AVX1-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
+; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP4]], align 1
+; AVX1-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX1-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
+; AVX1-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
+; AVX1-NEXT: [[TMP8:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX1-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
+; AVX1-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
+; AVX1-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX1-NEXT: [[TMP13:%.*]] = icmp ne <4 x i8> [[TMP8]], zeroinitializer
; AVX1-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
-; AVX1-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX1-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
+; AVX1-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
+; AVX1-NEXT: [[TMP16:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
+; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP9]], <4 x ptr> poison)
+; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
+; AVX1-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
+; AVX1-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP16]], <4 x i1> [[TMP13]], <4 x ptr> poison)
+; AVX1-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX1-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX1-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
+; AVX1-NEXT: [[TMP20:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
+; AVX1-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP25:%.*]] = select <4 x i1> [[TMP13]], <4 x i1> [[TMP20]], <4 x i1> zeroinitializer
; AVX1-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX1-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
-; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX1-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
+; AVX1-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
+; AVX1-NEXT: [[TMP30:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP21]])
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP30]], <4 x i1> [[TMP25]])
+; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX1-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP28:![0-9]+]]
; AVX1: [[MIDDLE_BLOCK]]:
; AVX1-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX1: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
+; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
; AVX1-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF29:![0-9]+]]
; AVX1: [[VEC_EPILOG_PH]]:
; AVX1-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
+; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
; AVX1-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX1-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX1: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
-; AVX1-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX1-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
+; AVX1-NEXT: [[WIDE_LOAD10:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
+; AVX1-NEXT: [[TMP34:%.*]] = and <4 x i8> [[WIDE_LOAD10]], splat (i8 1)
+; AVX1-NEXT: [[TMP33:%.*]] = icmp ne <4 x i8> [[TMP34]], zeroinitializer
; AVX1-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
-; AVX1-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX1-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
+; AVX1-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP33]], <4 x ptr> poison)
+; AVX1-NEXT: [[TMP37:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD11]], splat (ptr null)
+; AVX1-NEXT: [[TMP38:%.*]] = select <4 x i1> [[TMP33]], <4 x i1> [[TMP37]], <4 x i1> zeroinitializer
; AVX1-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX1-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
-; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP38]])
+; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
; AVX1-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
; AVX1-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP30:![0-9]+]]
; AVX1: [[VEC_EPILOG_MIDDLE_BLOCK]]:
@@ -1320,56 +1437,86 @@ define void @foo7(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX2-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX2: [[ITER_CHECK]]:
; AVX2-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
+; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX2: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
+; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX2: [[VECTOR_PH]]:
-; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
+; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
; AVX2-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX2-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX2: [[VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
-; AVX2-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX2-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
+; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
+; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
+; AVX2-NEXT: [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
+; AVX2-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
+; AVX2-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
+; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP4]], align 1
+; AVX2-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX2-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
+; AVX2-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
+; AVX2-NEXT: [[TMP8:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX2-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
+; AVX2-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
+; AVX2-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX2-NEXT: [[TMP13:%.*]] = icmp ne <4 x i8> [[TMP8]], zeroinitializer
; AVX2-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
-; AVX2-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX2-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
+; AVX2-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
+; AVX2-NEXT: [[TMP16:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP9]], <4 x ptr> poison)
+; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
+; AVX2-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
+; AVX2-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP16]], <4 x i1> [[TMP13]], <4 x ptr> poison)
+; AVX2-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX2-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX2-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
+; AVX2-NEXT: [[TMP20:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
+; AVX2-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP25:%.*]] = select <4 x i1> [[TMP13]], <4 x i1> [[TMP20]], <4 x i1> zeroinitializer
; AVX2-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX2-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX2-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
+; AVX2-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
+; AVX2-NEXT: [[TMP30:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP21]])
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP30]], <4 x i1> [[TMP25]])
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX2-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX2-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX2-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX2: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
-; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
+; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
+; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF33:![0-9]+]]
; AVX2: [[VEC_EPILOG_PH]]:
; AVX2-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
+; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
; AVX2-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX2-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX2: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
-; AVX2-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX2-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
+; AVX2-NEXT: [[WIDE_LOAD10:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
+; AVX2-NEXT: [[TMP34:%.*]] = and <4 x i8> [[WIDE_LOAD10]], splat (i8 1)
+; AVX2-NEXT: [[TMP33:%.*]] = icmp ne <4 x i8> [[TMP34]], zeroinitializer
; AVX2-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
-; AVX2-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX2-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
+; AVX2-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP33]], <4 x ptr> poison)
+; AVX2-NEXT: [[TMP37:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD11]], splat (ptr null)
+; AVX2-NEXT: [[TMP38:%.*]] = select <4 x i1> [[TMP33]], <4 x i1> [[TMP37]], <4 x i1> zeroinitializer
; AVX2-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
-; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP38]])
+; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
; AVX2-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
-; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP34:![0-9]+]]
; AVX2: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N14:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC9]]
; AVX2-NEXT: br i1 [[CMP_N14]], [[FOR_END_LOOPEXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
@@ -1405,13 +1552,13 @@ define void @foo7(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX512-NEXT: call void @llvm.masked.store.v64f64.p0(<64 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <64 x i1> [[TMP6]])
; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
; AVX512-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; AVX512-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP43:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP44:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX512-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX512: [[VEC_EPILOG_ITER_CHECK]]:
; AVX512-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
-; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF44:![0-9]+]]
+; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF45:![0-9]+]]
; AVX512: [[VEC_EPILOG_PH]]:
; AVX512-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX512-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
@@ -1431,7 +1578,7 @@ define void @foo7(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP34]])
; AVX512-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
; AVX512-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
-; AVX512-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP45:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP46:![0-9]+]]
; AVX512: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX512-NEXT: [[CMP_N14:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC9]]
; AVX512-NEXT: br i1 [[CMP_N14]], [[FOR_END_LOOPEXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
@@ -1489,54 +1636,84 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX1-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX1: [[ITER_CHECK]]:
; AVX1-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
+; AVX1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX1: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
+; AVX1-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
; AVX1-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX1: [[VECTOR_PH]]:
-; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
+; AVX1-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
; AVX1-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX1-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX1: [[VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
-; AVX1-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX1-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
+; AVX1-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
+; AVX1-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
+; AVX1-NEXT: [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
+; AVX1-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
+; AVX1-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
+; AVX1-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
+; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP4]], align 1
+; AVX1-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX1-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
+; AVX1-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
+; AVX1-NEXT: [[TMP8:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX1-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
+; AVX1-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
+; AVX1-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX1-NEXT: [[TMP13:%.*]] = icmp ne <4 x i8> [[TMP8]], zeroinitializer
; AVX1-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
-; AVX1-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX1-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
+; AVX1-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
+; AVX1-NEXT: [[TMP16:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
+; AVX1-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP9]], <4 x ptr> poison)
+; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
+; AVX1-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
+; AVX1-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP16]], <4 x i1> [[TMP13]], <4 x ptr> poison)
+; AVX1-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX1-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX1-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
+; AVX1-NEXT: [[TMP20:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
+; AVX1-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX1-NEXT: [[TMP25:%.*]] = select <4 x i1> [[TMP13]], <4 x i1> [[TMP20]], <4 x i1> zeroinitializer
; AVX1-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX1-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
-; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX1-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
+; AVX1-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
+; AVX1-NEXT: [[TMP30:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP21]])
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP30]], <4 x i1> [[TMP25]])
+; AVX1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX1-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
; AVX1: [[MIDDLE_BLOCK]]:
; AVX1-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX1-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX1: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
+; AVX1-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
; AVX1-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF29]]
; AVX1: [[VEC_EPILOG_PH]]:
; AVX1-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
+; AVX1-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
; AVX1-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX1-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX1: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX1-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX1-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
-; AVX1-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX1-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
+; AVX1-NEXT: [[WIDE_LOAD10:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
+; AVX1-NEXT: [[TMP34:%.*]] = and <4 x i8> [[WIDE_LOAD10]], splat (i8 1)
+; AVX1-NEXT: [[TMP33:%.*]] = icmp ne <4 x i8> [[TMP34]], zeroinitializer
; AVX1-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX1-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
-; AVX1-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX1-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
+; AVX1-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP33]], <4 x ptr> poison)
+; AVX1-NEXT: [[TMP37:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD11]], splat (ptr null)
+; AVX1-NEXT: [[TMP38:%.*]] = select <4 x i1> [[TMP33]], <4 x i1> [[TMP37]], <4 x i1> zeroinitializer
; AVX1-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX1-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
-; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
+; AVX1-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP38]])
+; AVX1-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
; AVX1-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
; AVX1-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP33:![0-9]+]]
; AVX1: [[VEC_EPILOG_MIDDLE_BLOCK]]:
@@ -1551,56 +1728,86 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX2-NEXT: br i1 [[CMP5]], [[FOR_END:label %.*]], label %[[ITER_CHECK:.*]]
; AVX2: [[ITER_CHECK]]:
; AVX2-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext i32 [[SIZE]] to i64
-; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 8
+; AVX2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 4
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
; AVX2: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 32
+; AVX2-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[WIDE_TRIP_COUNT]], 16
; AVX2-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; AVX2: [[VECTOR_PH]]:
-; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 31
+; AVX2-NEXT: [[N_MOD_VF:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 15
; AVX2-NEXT: [[N_VEC:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF]]
; AVX2-NEXT: br label %[[VECTOR_BODY:.*]]
; AVX2: [[VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX]]
-; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, ptr [[TMP0]], align 1
-; AVX2-NEXT: [[TMP2:%.*]] = and <32 x i8> [[WIDE_LOAD]], splat (i8 1)
-; AVX2-NEXT: [[TMP3:%.*]] = icmp ne <32 x i8> [[TMP2]], zeroinitializer
+; AVX2-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 4
+; AVX2-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 8
+; AVX2-NEXT: [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 12
+; AVX2-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
+; AVX2-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
+; AVX2-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
+; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i8>, ptr [[TMP4]], align 1
+; AVX2-NEXT: [[TMP5:%.*]] = and <4 x i8> [[WIDE_LOAD]], splat (i8 1)
+; AVX2-NEXT: [[TMP6:%.*]] = and <4 x i8> [[WIDE_LOAD2]], splat (i8 1)
+; AVX2-NEXT: [[TMP7:%.*]] = and <4 x i8> [[WIDE_LOAD3]], splat (i8 1)
+; AVX2-NEXT: [[TMP8:%.*]] = and <4 x i8> [[WIDE_LOAD4]], splat (i8 1)
+; AVX2-NEXT: [[TMP9:%.*]] = icmp ne <4 x i8> [[TMP5]], zeroinitializer
+; AVX2-NEXT: [[TMP10:%.*]] = icmp ne <4 x i8> [[TMP6]], zeroinitializer
+; AVX2-NEXT: [[TMP11:%.*]] = icmp ne <4 x i8> [[TMP7]], zeroinitializer
+; AVX2-NEXT: [[TMP13:%.*]] = icmp ne <4 x i8> [[TMP8]], zeroinitializer
; AVX2-NEXT: [[TMP12:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <32 x ptr> @llvm.masked.load.v32p0.p0(ptr align 8 [[TMP12]], <32 x i1> [[TMP3]], <32 x ptr> poison)
-; AVX2-NEXT: [[TMP5:%.*]] = icmp ne <32 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
-; AVX2-NEXT: [[TMP6:%.*]] = select <32 x i1> [[TMP3]], <32 x i1> [[TMP5]], <32 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP14:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 4
+; AVX2-NEXT: [[TMP15:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 8
+; AVX2-NEXT: [[TMP16:%.*]] = getelementptr ptr, ptr [[TMP12]], i64 12
+; AVX2-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP12]], <4 x i1> [[TMP9]], <4 x ptr> poison)
+; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP14]], <4 x i1> [[TMP10]], <4 x ptr> poison)
+; AVX2-NEXT: [[WIDE_MASKED_LOAD6:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP15]], <4 x i1> [[TMP11]], <4 x ptr> poison)
+; AVX2-NEXT: [[WIDE_MASKED_LOAD7:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP16]], <4 x i1> [[TMP13]], <4 x ptr> poison)
+; AVX2-NEXT: [[TMP17:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD]], splat (ptr null)
+; AVX2-NEXT: [[TMP18:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
+; AVX2-NEXT: [[TMP19:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD6]], splat (ptr null)
+; AVX2-NEXT: [[TMP20:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD7]], splat (ptr null)
+; AVX2-NEXT: [[TMP21:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP17]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP18]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP23:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP19]], <4 x i1> zeroinitializer
+; AVX2-NEXT: [[TMP25:%.*]] = select <4 x i1> [[TMP13]], <4 x i1> [[TMP20]], <4 x i1> zeroinitializer
; AVX2-NEXT: [[TMP24:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX]]
-; AVX2-NEXT: call void @llvm.masked.store.v32f64.p0(<32 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <32 x i1> [[TMP6]])
-; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; AVX2-NEXT: [[TMP26:%.*]] = getelementptr double, ptr [[TMP24]], i64 4
+; AVX2-NEXT: [[TMP27:%.*]] = getelementptr double, ptr [[TMP24]], i64 8
+; AVX2-NEXT: [[TMP30:%.*]] = getelementptr double, ptr [[TMP24]], i64 12
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <4 x i1> [[TMP21]])
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP26]], <4 x i1> [[TMP22]])
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP27]], <4 x i1> [[TMP23]])
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP30]], <4 x i1> [[TMP25]])
+; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
; AVX2-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; AVX2-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP35:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX2-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX2: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
-; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
+; AVX2-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 4
+; AVX2-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF33]]
; AVX2: [[VEC_EPILOG_PH]]:
; AVX2-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
+; AVX2-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 3
; AVX2-NEXT: [[N_VEC9:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[N_MOD_VF8]]
; AVX2-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; AVX2: [[VEC_EPILOG_VECTOR_BODY]]:
; AVX2-NEXT: [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
; AVX2-NEXT: [[TMP29:%.*]] = getelementptr inbounds i8, ptr [[TRIGGER]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_LOAD4:%.*]] = load <8 x i8>, ptr [[TMP29]], align 1
-; AVX2-NEXT: [[TMP11:%.*]] = and <8 x i8> [[WIDE_LOAD4]], splat (i8 1)
-; AVX2-NEXT: [[TMP13:%.*]] = icmp ne <8 x i8> [[TMP11]], zeroinitializer
+; AVX2-NEXT: [[WIDE_LOAD10:%.*]] = load <4 x i8>, ptr [[TMP29]], align 1
+; AVX2-NEXT: [[TMP34:%.*]] = and <4 x i8> [[WIDE_LOAD10]], splat (i8 1)
+; AVX2-NEXT: [[TMP33:%.*]] = icmp ne <4 x i8> [[TMP34]], zeroinitializer
; AVX2-NEXT: [[TMP32:%.*]] = getelementptr ptr, ptr [[IN]], i64 [[INDEX10]]
-; AVX2-NEXT: [[WIDE_MASKED_LOAD5:%.*]] = call <8 x ptr> @llvm.masked.load.v8p0.p0(ptr align 8 [[TMP32]], <8 x i1> [[TMP13]], <8 x ptr> poison)
-; AVX2-NEXT: [[TMP14:%.*]] = icmp ne <8 x ptr> [[WIDE_MASKED_LOAD5]], splat (ptr null)
-; AVX2-NEXT: [[TMP15:%.*]] = select <8 x i1> [[TMP13]], <8 x i1> [[TMP14]], <8 x i1> zeroinitializer
+; AVX2-NEXT: [[WIDE_MASKED_LOAD11:%.*]] = call <4 x ptr> @llvm.masked.load.v4p0.p0(ptr align 8 [[TMP32]], <4 x i1> [[TMP33]], <4 x ptr> poison)
+; AVX2-NEXT: [[TMP37:%.*]] = icmp ne <4 x ptr> [[WIDE_MASKED_LOAD11]], splat (ptr null)
+; AVX2-NEXT: [[TMP38:%.*]] = select <4 x i1> [[TMP33]], <4 x i1> [[TMP37]], <4 x i1> zeroinitializer
; AVX2-NEXT: [[TMP35:%.*]] = getelementptr double, ptr [[OUT]], i64 [[INDEX10]]
-; AVX2-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP15]])
-; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
+; AVX2-NEXT: call void @llvm.masked.store.v4f64.p0(<4 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <4 x i1> [[TMP38]])
+; AVX2-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 4
; AVX2-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
-; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP36:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP37:![0-9]+]]
; AVX2: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX2-NEXT: [[CMP_N14:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC9]]
; AVX2-NEXT: br i1 [[CMP_N14]], [[FOR_END_LOOPEXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
@@ -1636,13 +1843,13 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX512-NEXT: call void @llvm.masked.store.v64f64.p0(<64 x double> splat (double 5.000000e-01), ptr align 8 [[TMP24]], <64 x i1> [[TMP6]])
; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 64
; AVX512-NEXT: [[TMP28:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; AVX512-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP47:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP28]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP48:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC]]
; AVX512-NEXT: br i1 [[CMP_N]], [[FOR_END_LOOPEXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX512: [[VEC_EPILOG_ITER_CHECK]]:
; AVX512-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], 8
-; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF44]]
+; AVX512-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF45]]
; AVX512: [[VEC_EPILOG_PH]]:
; AVX512-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX512-NEXT: [[N_MOD_VF8:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 7
@@ -1662,7 +1869,7 @@ define void @foo8(ptr noalias nocapture %out, ptr noalias nocapture readonly %in
; AVX512-NEXT: call void @llvm.masked.store.v8f64.p0(<8 x double> splat (double 5.000000e-01), ptr align 8 [[TMP35]], <8 x i1> [[TMP34]])
; AVX512-NEXT: [[INDEX_NEXT13]] = add nuw i64 [[INDEX10]], 8
; AVX512-NEXT: [[TMP36:%.*]] = icmp eq i64 [[INDEX_NEXT13]], [[N_VEC9]]
-; AVX512-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP48:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP36]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP49:![0-9]+]]
; AVX512: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX512-NEXT: [[CMP_N14:%.*]] = icmp eq i64 [[WIDE_TRIP_COUNT]], [[N_VEC9]]
; AVX512-NEXT: br i1 [[CMP_N14]], [[FOR_END_LOOPEXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
@@ -1941,7 +2148,7 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: [[TMP199]] = or <8 x i1> [[VEC_PHI3]], [[TMP195]]
; AVX2-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 32
; AVX2-NEXT: [[TMP104:%.*]] = icmp eq i32 [[INDEX_NEXT]], 96
-; AVX2-NEXT: br i1 [[TMP104]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP38:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP104]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP39:![0-9]+]]
; AVX2: [[MIDDLE_BLOCK]]:
; AVX2-NEXT: [[BIN_RDX:%.*]] = or <8 x i1> [[TMP197]], [[TMP196]]
; AVX2-NEXT: [[BIN_RDX4:%.*]] = or <8 x i1> [[TMP198]], [[BIN_RDX]]
@@ -1951,7 +2158,7 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: [[RDX_SELECT:%.*]] = select i1 [[TMP106]], i32 0, i32 1
; AVX2-NEXT: br i1 false, [[EXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX2: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX2-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF39:![0-9]+]]
+; AVX2-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF40:![0-9]+]]
; AVX2: [[VEC_EPILOG_PH]]:
; AVX2-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ 96, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX2-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 1, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -1990,7 +2197,7 @@ define i32 @reverse_gather(ptr %p) {
; AVX2-NEXT: [[TMP133]] = or <4 x i1> [[VEC_PHI7]], [[TMP132]]
; AVX2-NEXT: [[INDEX_NEXT8]] = add nuw i32 [[INDEX6]], 4
; AVX2-NEXT: [[TMP134:%.*]] = icmp eq i32 [[INDEX_NEXT8]], 100
-; AVX2-NEXT: br i1 [[TMP134]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP40:![0-9]+]]
+; AVX2-NEXT: br i1 [[TMP134]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP41:![0-9]+]]
; AVX2: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX2-NEXT: [[TMP135:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP133]])
; AVX2-NEXT: [[TMP136:%.*]] = freeze i1 [[TMP135]]
@@ -2041,7 +2248,7 @@ define i32 @reverse_gather(ptr %p) {
; AVX512-NEXT: [[TMP14]] = or <8 x i1> [[VEC_PHI3]], [[TMP10]]
; AVX512-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 32
; AVX512-NEXT: [[TMP15:%.*]] = icmp eq i32 [[INDEX_NEXT]], 96
-; AVX512-NEXT: br i1 [[TMP15]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP50:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP15]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP51:![0-9]+]]
; AVX512: [[MIDDLE_BLOCK]]:
; AVX512-NEXT: [[BIN_RDX:%.*]] = or <8 x i1> [[TMP12]], [[TMP11]]
; AVX512-NEXT: [[BIN_RDX13:%.*]] = or <8 x i1> [[TMP13]], [[BIN_RDX]]
@@ -2051,7 +2258,7 @@ define i32 @reverse_gather(ptr %p) {
; AVX512-NEXT: [[RDX_SELECT:%.*]] = select i1 [[TMP17]], i32 0, i32 1
; AVX512-NEXT: br i1 false, [[EXIT:label %.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
; AVX512: [[VEC_EPILOG_ITER_CHECK]]:
-; AVX512-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF51:![0-9]+]]
+; AVX512-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF52:![0-9]+]]
; AVX512: [[VEC_EPILOG_PH]]:
; AVX512-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ 96, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
; AVX512-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 1, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -2084,7 +2291,7 @@ define i32 @reverse_gather(ptr %p) {
; AVX512-NEXT: [[TMP36]] = or <4 x i1> [[VEC_PHI16]], [[TMP35]]
; AVX512-NEXT: [[INDEX_NEXT19]] = add nuw i32 [[INDEX15]], 4
; AVX512-NEXT: [[TMP37:%.*]] = icmp eq i32 [[INDEX_NEXT19]], 100
-; AVX512-NEXT: br i1 [[TMP37]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP52:![0-9]+]]
+; AVX512-NEXT: br i1 [[TMP37]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP53:![0-9]+]]
; AVX512: [[VEC_EPILOG_MIDDLE_BLOCK]]:
; AVX512-NEXT: [[TMP38:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP36]])
; AVX512-NEXT: [[TMP39:%.*]] = freeze i1 [[TMP38]]
More information about the llvm-commits
mailing list