[llvm] [LV] Support interleaving in order to reduce stalls (PR #197959)

John Brawn via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 11 07:16:04 PDT 2026


https://github.com/john-brawn-arm updated https://github.com/llvm/llvm-project/pull/197959

>From 90d56622807fdc0f6776b4e6f9d41e7f3e9a0e3b Mon Sep 17 00:00:00 2001
From: John Brawn <john.brawn at arm.com>
Date: Fri, 17 Apr 2026 11:02:13 +0100
Subject: [PATCH] [LV] Support interleaving in order to reduce stalls

On an in-order CPU the CPU will stall after a load if the value is
used immediately. Interleaving such that the combined reciprocal
throughput of all of the load instructions is equal to the load
latency means we will avoid this stall.
---
 .../llvm/Analysis/TargetTransformInfo.h       |    4 +
 .../llvm/Analysis/TargetTransformInfoImpl.h   |    2 +
 llvm/include/llvm/CodeGen/BasicTTIImpl.h      |    2 +
 llvm/lib/Analysis/TargetTransformInfo.cpp     |    4 +
 .../Transforms/Vectorize/LoopVectorize.cpp    |   48 +
 .../AArch64/interleave-for-latency.ll         | 1198 +++++++++++++++++
 6 files changed, 1258 insertions(+)
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/interleave-for-latency.ll

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index 10c0509460b95..6f7e42d769b04 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -1504,6 +1504,10 @@ class TargetTransformInfo {
   LLVM_ABI unsigned getMaxInterleaveFactor(ElementCount VF,
                                            bool HasUnorderedReductions) const;
 
+  /// \return True if the loop vectorizer should interleave in order to reduce
+  /// the number of stall cycles due to long latency instructions.
+  LLVM_ABI bool shouldInterleaveToReduceStalls() const;
+
   /// Collect properties of V used in cost analysis, e.g. OP_PowerOf2.
   LLVM_ABI static OperandValueInfo getOperandInfo(const Value *V);
 
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index 5645cc63a6944..6dec433bb7f73 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -727,6 +727,8 @@ class LLVM_ABI TargetTransformInfoImplBase {
     return 1;
   }
 
+  virtual bool shouldInterleaveToReduceStalls() const { return false; }
+
   virtual InstructionCost getArithmeticInstrCost(
       unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
       TTI::OperandValueInfo Opd1Info, TTI::OperandValueInfo Opd2Info,
diff --git a/llvm/include/llvm/CodeGen/BasicTTIImpl.h b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
index 46e18593430a2..09defd58e7227 100644
--- a/llvm/include/llvm/CodeGen/BasicTTIImpl.h
+++ b/llvm/include/llvm/CodeGen/BasicTTIImpl.h
@@ -1061,6 +1061,8 @@ class BasicTTIImplBase : public TargetTransformInfoImplCRTPBase<T> {
     return 1;
   }
 
+  bool shouldInterleaveToReduceStalls() const override { return false; }
+
   InstructionCost getArithmeticInstrCost(
       unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
       TTI::OperandValueInfo Opd1Info = {TTI::OK_AnyValue, TTI::OP_None},
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index cdccd04f4c9ea..bfd9a3923c26e 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -932,6 +932,10 @@ TargetTransformInfo::getMaxInterleaveFactor(ElementCount VF,
   return TTIImpl->getMaxInterleaveFactor(VF, HasUnorderedReductions);
 }
 
+bool TargetTransformInfo::shouldInterleaveToReduceStalls() const {
+  return TTIImpl->shouldInterleaveToReduceStalls();
+}
+
 TargetTransformInfo::OperandValueInfo
 TargetTransformInfo::getOperandInfo(const Value *V) {
   OperandValueKind OpInfo = OK_AnyValue;
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 6bb68d1f7bb19..2e36b5f890035 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -287,12 +287,22 @@ static cl::opt<unsigned> ForceTargetMaxVectorInterleaveFactor(
     cl::desc("A flag that overrides the target's max interleave factor for "
              "vectorized loops."));
 
+static cl::opt<bool> EnableInterleaveToReduceStalls(
+    "enable-interleave-to-reduce-stalls", cl::init(false), cl::Hidden,
+    cl::desc("When interleaving choose the interleave count that will reduce "
+             "the stall cycles due to instruction latency to zero."));
+
 cl::opt<unsigned> llvm::ForceTargetInstructionCost(
     "force-target-instruction-cost", cl::init(0), cl::Hidden,
     cl::desc("A flag that overrides the target's expected cost for "
              "an instruction to a single constant value. Mostly "
              "useful for getting consistent testing."));
 
+static cl::opt<unsigned>
+    ForceTargetLoadLatency("force-target-load-latency", cl::init(0), cl::Hidden,
+                           cl::desc("A flag that overrides the target's "
+                                    "expected latency for load instructions."));
+
 static cl::opt<unsigned> SmallLoopCost(
     "small-loop-cost", cl::init(20), cl::Hidden,
     cl::desc(
@@ -3840,6 +3850,44 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
 
   assert(IC > 0 && "Interleave count must be greater than 0.");
 
+  if (TTI.shouldInterleaveToReduceStalls() || EnableInterleaveToReduceStalls) {
+    LLVM_DEBUG(dbgs() << "LV: Interleaving to reduce stall cycles due to "
+                         "instruction latency.\n");
+    VPCostContext LatencyCtx(CM.TTI, *CM.TLI, Plan, CM, TTI::TCK_Latency, PSE,
+                             OrigLoop);
+    VPCostContext ThroughputCtx(CM.TTI, *CM.TLI, Plan, CM,
+                                TTI::TCK_RecipThroughput, PSE, OrigLoop);
+    // Find the largest interleave count that will help with reducing stalls.
+    unsigned StallsIC = 1;
+    for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
+             vp_depth_first_shallow(Plan.getVectorLoopRegion()->getEntry()))) {
+      for (VPRecipeBase &R : *VPBB) {
+        // Assuming that each value will be needed as soon as it's generated the
+        // number of stall cycles is one less than the latency.
+        InstructionCost Stalls = R.cost(VF, LatencyCtx) - 1;
+        if (ForceTargetLoadLatency.getNumOccurrences() > 0 &&
+            R.mayReadFromMemory())
+          Stalls = ForceTargetLoadLatency - 1;
+        // Each interleaving above 1 will reduce the stalls by RecipThroughput,
+        // so pick the interleaving that will reduce stalls to zero.
+        InstructionCost RecipThroughput = R.cost(VF, ThroughputCtx);
+        if (Stalls.isValid() && RecipThroughput.isValid() && Stalls > 0 &&
+            RecipThroughput > 0) {
+          unsigned ThisIC =
+              bit_floor<uint64_t>(1 + (Stalls / RecipThroughput).getValue());
+          StallsIC = std::max(StallsIC, ThisIC);
+        }
+      }
+    }
+    if (StallsIC <= 1) {
+      LLVM_DEBUG(
+          dbgs() << "LV: Not interleaving as it wouldn't reduce stalls.\n");
+      return 1;
+    }
+    IC = std::min(IC, StallsIC);
+    return IC;
+  }
+
   // Interleave if we vectorized this loop and there is a reduction that could
   // benefit from interleaving.
   if (VF.isVector() && HasReductions) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/interleave-for-latency.ll b/llvm/test/Transforms/LoopVectorize/AArch64/interleave-for-latency.ll
new file mode 100644
index 0000000000000..4126efe09a4de
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/interleave-for-latency.ll
@@ -0,0 +1,1198 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -passes=loop-vectorize -enable-interleave-to-reduce-stalls -force-target-max-vector-interleave=4 -force-target-load-latency=1 -S %s | FileCheck %s --check-prefixes=CHECK-LATENCY1
+; RUN: opt -passes=loop-vectorize -enable-interleave-to-reduce-stalls -force-target-max-vector-interleave=4 -force-target-load-latency=2 -S %s | FileCheck %s --check-prefixes=CHECK-LATENCY2
+; RUN: opt -passes=loop-vectorize -enable-interleave-to-reduce-stalls -force-target-max-vector-interleave=4 -force-target-load-latency=8 -S %s | FileCheck %s --check-prefixes=CHECK-LATENCY8
+; RUN: opt -passes=loop-vectorize -enable-interleave-to-reduce-stalls -force-target-max-vector-interleave=4 -mcpu=cortex-a510 -S %s | FileCheck %s --check-prefixes=CHECK-A510
+; RUN: opt -passes=loop-vectorize -enable-interleave-to-reduce-stalls -force-target-max-vector-interleave=4 -mcpu=cortex-a320 -S %s | FileCheck %s --check-prefixes=CHECK-A320
+target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128-Fn32"
+target triple = "aarch64-unknown-none-elf"
+
+; In these tests we expect that the loop is interleaved enough that we have as
+; many loads as the load latency. The exception is for latency 8, due to the
+; max interleave factor being 4.
+
+define void @i16_add1(ptr noalias readonly %src, ptr noalias writeonly %dst, i16 %argval) {
+; CHECK-LATENCY1-LABEL: define void @i16_add1(
+; CHECK-LATENCY1-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY1-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY1-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY1:       [[VECTOR_PH]]:
+; CHECK-LATENCY1-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-LATENCY1-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-LATENCY1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY1:       [[VECTOR_BODY]]:
+; CHECK-LATENCY1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY1-NEXT:    [[TMP0:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-LATENCY1-NEXT:    [[WIDE_LOAD:%.*]] = load <8 x i16>, ptr [[TMP0]], align 2
+; CHECK-LATENCY1-NEXT:    [[TMP1:%.*]] = add <8 x i16> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY1-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[INDEX]]
+; CHECK-LATENCY1-NEXT:    store <8 x i16> [[TMP1]], ptr [[TMP2]], align 2
+; CHECK-LATENCY1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-LATENCY1-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-LATENCY1-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-LATENCY1:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY1-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY1:       [[EXIT]]:
+; CHECK-LATENCY1-NEXT:    ret void
+;
+; CHECK-LATENCY2-LABEL: define void @i16_add1(
+; CHECK-LATENCY2-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY2-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY2-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY2:       [[VECTOR_PH]]:
+; CHECK-LATENCY2-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-LATENCY2-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-LATENCY2-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY2:       [[VECTOR_BODY]]:
+; CHECK-LATENCY2-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY2-NEXT:    [[TMP0:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-LATENCY2-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP0]], i64 8
+; CHECK-LATENCY2-NEXT:    [[WIDE_LOAD:%.*]] = load <8 x i16>, ptr [[TMP0]], align 2
+; CHECK-LATENCY2-NEXT:    [[WIDE_LOAD1:%.*]] = load <8 x i16>, ptr [[TMP1]], align 2
+; CHECK-LATENCY2-NEXT:    [[TMP2:%.*]] = add <8 x i16> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY2-NEXT:    [[TMP3:%.*]] = add <8 x i16> [[WIDE_LOAD1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY2-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[INDEX]]
+; CHECK-LATENCY2-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP4]], i64 8
+; CHECK-LATENCY2-NEXT:    store <8 x i16> [[TMP2]], ptr [[TMP4]], align 2
+; CHECK-LATENCY2-NEXT:    store <8 x i16> [[TMP3]], ptr [[TMP5]], align 2
+; CHECK-LATENCY2-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-LATENCY2-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-LATENCY2-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-LATENCY2:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY2-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY2:       [[EXIT]]:
+; CHECK-LATENCY2-NEXT:    ret void
+;
+; CHECK-LATENCY8-LABEL: define void @i16_add1(
+; CHECK-LATENCY8-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY8-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY8-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY8:       [[VECTOR_PH]]:
+; CHECK-LATENCY8-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-LATENCY8-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-LATENCY8-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY8:       [[VECTOR_BODY]]:
+; CHECK-LATENCY8-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY8-NEXT:    [[TMP0:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-LATENCY8-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP0]], i64 8
+; CHECK-LATENCY8-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP0]], i64 16
+; CHECK-LATENCY8-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP0]], i64 24
+; CHECK-LATENCY8-NEXT:    [[WIDE_LOAD:%.*]] = load <8 x i16>, ptr [[TMP0]], align 2
+; CHECK-LATENCY8-NEXT:    [[WIDE_LOAD1:%.*]] = load <8 x i16>, ptr [[TMP1]], align 2
+; CHECK-LATENCY8-NEXT:    [[WIDE_LOAD2:%.*]] = load <8 x i16>, ptr [[TMP2]], align 2
+; CHECK-LATENCY8-NEXT:    [[WIDE_LOAD3:%.*]] = load <8 x i16>, ptr [[TMP3]], align 2
+; CHECK-LATENCY8-NEXT:    [[TMP4:%.*]] = add <8 x i16> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP5:%.*]] = add <8 x i16> [[WIDE_LOAD1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP6:%.*]] = add <8 x i16> [[WIDE_LOAD2]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP7:%.*]] = add <8 x i16> [[WIDE_LOAD3]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP8:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[INDEX]]
+; CHECK-LATENCY8-NEXT:    [[TMP9:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP8]], i64 8
+; CHECK-LATENCY8-NEXT:    [[TMP10:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP8]], i64 16
+; CHECK-LATENCY8-NEXT:    [[TMP11:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP8]], i64 24
+; CHECK-LATENCY8-NEXT:    store <8 x i16> [[TMP4]], ptr [[TMP8]], align 2
+; CHECK-LATENCY8-NEXT:    store <8 x i16> [[TMP5]], ptr [[TMP9]], align 2
+; CHECK-LATENCY8-NEXT:    store <8 x i16> [[TMP6]], ptr [[TMP10]], align 2
+; CHECK-LATENCY8-NEXT:    store <8 x i16> [[TMP7]], ptr [[TMP11]], align 2
+; CHECK-LATENCY8-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-LATENCY8-NEXT:    [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-LATENCY8-NEXT:    br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-LATENCY8:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY8-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY8:       [[EXIT]]:
+; CHECK-LATENCY8-NEXT:    ret void
+;
+; CHECK-A510-LABEL: define void @i16_add1(
+; CHECK-A510-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-A510-NEXT:  [[ENTRY:.*:]]
+; CHECK-A510-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-A510:       [[VECTOR_PH]]:
+; CHECK-A510-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-A510-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-A510-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-A510:       [[VECTOR_BODY]]:
+; CHECK-A510-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-A510-NEXT:    [[TMP0:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-A510-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP0]], i64 8
+; CHECK-A510-NEXT:    [[WIDE_LOAD:%.*]] = load <8 x i16>, ptr [[TMP0]], align 2
+; CHECK-A510-NEXT:    [[WIDE_LOAD1:%.*]] = load <8 x i16>, ptr [[TMP1]], align 2
+; CHECK-A510-NEXT:    [[TMP2:%.*]] = add <8 x i16> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-A510-NEXT:    [[TMP3:%.*]] = add <8 x i16> [[WIDE_LOAD1]], [[BROADCAST_SPLAT]]
+; CHECK-A510-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[INDEX]]
+; CHECK-A510-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP4]], i64 8
+; CHECK-A510-NEXT:    store <8 x i16> [[TMP2]], ptr [[TMP4]], align 2
+; CHECK-A510-NEXT:    store <8 x i16> [[TMP3]], ptr [[TMP5]], align 2
+; CHECK-A510-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-A510-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-A510-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-A510:       [[MIDDLE_BLOCK]]:
+; CHECK-A510-NEXT:    br label %[[EXIT:.*]]
+; CHECK-A510:       [[EXIT]]:
+; CHECK-A510-NEXT:    ret void
+;
+; CHECK-A320-LABEL: define void @i16_add1(
+; CHECK-A320-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-A320-NEXT:  [[ENTRY:.*:]]
+; CHECK-A320-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-A320:       [[VECTOR_PH]]:
+; CHECK-A320-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-A320-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-A320-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-A320:       [[VECTOR_BODY]]:
+; CHECK-A320-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-A320-NEXT:    [[TMP0:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-A320-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP0]], i64 8
+; CHECK-A320-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP0]], i64 16
+; CHECK-A320-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP0]], i64 24
+; CHECK-A320-NEXT:    [[WIDE_LOAD:%.*]] = load <8 x i16>, ptr [[TMP0]], align 2
+; CHECK-A320-NEXT:    [[WIDE_LOAD1:%.*]] = load <8 x i16>, ptr [[TMP1]], align 2
+; CHECK-A320-NEXT:    [[WIDE_LOAD2:%.*]] = load <8 x i16>, ptr [[TMP2]], align 2
+; CHECK-A320-NEXT:    [[WIDE_LOAD3:%.*]] = load <8 x i16>, ptr [[TMP3]], align 2
+; CHECK-A320-NEXT:    [[TMP4:%.*]] = add <8 x i16> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP5:%.*]] = add <8 x i16> [[WIDE_LOAD1]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP6:%.*]] = add <8 x i16> [[WIDE_LOAD2]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP7:%.*]] = add <8 x i16> [[WIDE_LOAD3]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP8:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[INDEX]]
+; CHECK-A320-NEXT:    [[TMP9:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP8]], i64 8
+; CHECK-A320-NEXT:    [[TMP10:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP8]], i64 16
+; CHECK-A320-NEXT:    [[TMP11:%.*]] = getelementptr inbounds nuw i16, ptr [[TMP8]], i64 24
+; CHECK-A320-NEXT:    store <8 x i16> [[TMP4]], ptr [[TMP8]], align 2
+; CHECK-A320-NEXT:    store <8 x i16> [[TMP5]], ptr [[TMP9]], align 2
+; CHECK-A320-NEXT:    store <8 x i16> [[TMP6]], ptr [[TMP10]], align 2
+; CHECK-A320-NEXT:    store <8 x i16> [[TMP7]], ptr [[TMP11]], align 2
+; CHECK-A320-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-A320-NEXT:    [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-A320-NEXT:    br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-A320:       [[MIDDLE_BLOCK]]:
+; CHECK-A320-NEXT:    br label %[[EXIT:.*]]
+; CHECK-A320:       [[EXIT]]:
+; CHECK-A320-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ %iv.next, %loop ], [ 0, %entry ]
+  %src0 = getelementptr inbounds nuw i16, ptr %src, i64 %iv
+  %val0 = load i16, ptr %src0, align 2
+  %add0 = add i16 %val0, %argval
+  %dst0 = getelementptr inbounds nuw i16, ptr %dst, i64 %iv
+  store i16 %add0, ptr %dst0, align 2
+  %iv.next = add nuw i64 %iv, 1
+  %cmp = icmp ult i64 %iv.next, 1024
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; FIXME: On a320 and a510 we should be interleaving by 2 here, but
+; latency of interleaved loads is not yet implemented.
+define void @i16_add2(ptr noalias readonly %src, ptr noalias writeonly %dst, i16 %argval) {
+; CHECK-LATENCY1-LABEL: define void @i16_add2(
+; CHECK-LATENCY1-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY1-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY1-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY1:       [[VECTOR_PH]]:
+; CHECK-LATENCY1-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-LATENCY1-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-LATENCY1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY1:       [[VECTOR_BODY]]:
+; CHECK-LATENCY1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY1-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 1
+; CHECK-LATENCY1-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-LATENCY1-NEXT:    [[WIDE_VEC:%.*]] = load <16 x i16>, ptr [[TMP1]], align 2
+; CHECK-LATENCY1-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <16 x i16> [[WIDE_VEC]], <16 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
+; CHECK-LATENCY1-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <16 x i16> [[WIDE_VEC]], <16 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
+; CHECK-LATENCY1-NEXT:    [[TMP2:%.*]] = add <8 x i16> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY1-NEXT:    [[TMP3:%.*]] = add <8 x i16> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY1-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[TMP0]]
+; CHECK-LATENCY1-NEXT:    [[TMP5:%.*]] = shufflevector <8 x i16> [[TMP2]], <8 x i16> [[TMP3]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-LATENCY1-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <16 x i16> [[TMP5]], <16 x i16> poison, <16 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11, i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
+; CHECK-LATENCY1-NEXT:    store <16 x i16> [[INTERLEAVED_VEC]], ptr [[TMP4]], align 2
+; CHECK-LATENCY1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-LATENCY1-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 512
+; CHECK-LATENCY1-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-LATENCY1:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY1-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY1:       [[EXIT]]:
+; CHECK-LATENCY1-NEXT:    ret void
+;
+; CHECK-LATENCY2-LABEL: define void @i16_add2(
+; CHECK-LATENCY2-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY2-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY2-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY2:       [[VECTOR_PH]]:
+; CHECK-LATENCY2-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-LATENCY2-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-LATENCY2-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY2:       [[VECTOR_BODY]]:
+; CHECK-LATENCY2-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY2-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 1
+; CHECK-LATENCY2-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-LATENCY2-NEXT:    [[WIDE_VEC:%.*]] = load <16 x i16>, ptr [[TMP1]], align 2
+; CHECK-LATENCY2-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <16 x i16> [[WIDE_VEC]], <16 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
+; CHECK-LATENCY2-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <16 x i16> [[WIDE_VEC]], <16 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
+; CHECK-LATENCY2-NEXT:    [[TMP2:%.*]] = add <8 x i16> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY2-NEXT:    [[TMP3:%.*]] = add <8 x i16> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY2-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[TMP0]]
+; CHECK-LATENCY2-NEXT:    [[TMP5:%.*]] = shufflevector <8 x i16> [[TMP2]], <8 x i16> [[TMP3]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-LATENCY2-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <16 x i16> [[TMP5]], <16 x i16> poison, <16 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11, i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
+; CHECK-LATENCY2-NEXT:    store <16 x i16> [[INTERLEAVED_VEC]], ptr [[TMP4]], align 2
+; CHECK-LATENCY2-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-LATENCY2-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 512
+; CHECK-LATENCY2-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-LATENCY2:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY2-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY2:       [[EXIT]]:
+; CHECK-LATENCY2-NEXT:    ret void
+;
+; CHECK-LATENCY8-LABEL: define void @i16_add2(
+; CHECK-LATENCY8-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY8-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY8-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY8:       [[VECTOR_PH]]:
+; CHECK-LATENCY8-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-LATENCY8-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-LATENCY8-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY8:       [[VECTOR_BODY]]:
+; CHECK-LATENCY8-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY8-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 1
+; CHECK-LATENCY8-NEXT:    [[TMP1:%.*]] = add i64 [[TMP0]], 16
+; CHECK-LATENCY8-NEXT:    [[TMP2:%.*]] = add i64 [[TMP0]], 32
+; CHECK-LATENCY8-NEXT:    [[TMP3:%.*]] = add i64 [[TMP0]], 48
+; CHECK-LATENCY8-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-LATENCY8-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[TMP1]]
+; CHECK-LATENCY8-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[TMP2]]
+; CHECK-LATENCY8-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[TMP3]]
+; CHECK-LATENCY8-NEXT:    [[WIDE_VEC:%.*]] = load <16 x i16>, ptr [[TMP4]], align 2
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <16 x i16> [[WIDE_VEC]], <16 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <16 x i16> [[WIDE_VEC]], <16 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
+; CHECK-LATENCY8-NEXT:    [[WIDE_VEC2:%.*]] = load <16 x i16>, ptr [[TMP5]], align 2
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC3:%.*]] = shufflevector <16 x i16> [[WIDE_VEC2]], <16 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC4:%.*]] = shufflevector <16 x i16> [[WIDE_VEC2]], <16 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
+; CHECK-LATENCY8-NEXT:    [[WIDE_VEC5:%.*]] = load <16 x i16>, ptr [[TMP6]], align 2
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC6:%.*]] = shufflevector <16 x i16> [[WIDE_VEC5]], <16 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC7:%.*]] = shufflevector <16 x i16> [[WIDE_VEC5]], <16 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
+; CHECK-LATENCY8-NEXT:    [[WIDE_VEC8:%.*]] = load <16 x i16>, ptr [[TMP7]], align 2
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC9:%.*]] = shufflevector <16 x i16> [[WIDE_VEC8]], <16 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC10:%.*]] = shufflevector <16 x i16> [[WIDE_VEC8]], <16 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
+; CHECK-LATENCY8-NEXT:    [[TMP8:%.*]] = add <8 x i16> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP9:%.*]] = add <8 x i16> [[STRIDED_VEC3]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP10:%.*]] = add <8 x i16> [[STRIDED_VEC6]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP11:%.*]] = add <8 x i16> [[STRIDED_VEC9]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP12:%.*]] = add <8 x i16> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP13:%.*]] = add <8 x i16> [[STRIDED_VEC4]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP14:%.*]] = add <8 x i16> [[STRIDED_VEC7]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP15:%.*]] = add <8 x i16> [[STRIDED_VEC10]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP16:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[TMP0]]
+; CHECK-LATENCY8-NEXT:    [[TMP17:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[TMP1]]
+; CHECK-LATENCY8-NEXT:    [[TMP18:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[TMP2]]
+; CHECK-LATENCY8-NEXT:    [[TMP19:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[TMP3]]
+; CHECK-LATENCY8-NEXT:    [[TMP20:%.*]] = shufflevector <8 x i16> [[TMP8]], <8 x i16> [[TMP12]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-LATENCY8-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <16 x i16> [[TMP20]], <16 x i16> poison, <16 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11, i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
+; CHECK-LATENCY8-NEXT:    store <16 x i16> [[INTERLEAVED_VEC]], ptr [[TMP16]], align 2
+; CHECK-LATENCY8-NEXT:    [[TMP21:%.*]] = shufflevector <8 x i16> [[TMP9]], <8 x i16> [[TMP13]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-LATENCY8-NEXT:    [[INTERLEAVED_VEC11:%.*]] = shufflevector <16 x i16> [[TMP21]], <16 x i16> poison, <16 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11, i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
+; CHECK-LATENCY8-NEXT:    store <16 x i16> [[INTERLEAVED_VEC11]], ptr [[TMP17]], align 2
+; CHECK-LATENCY8-NEXT:    [[TMP22:%.*]] = shufflevector <8 x i16> [[TMP10]], <8 x i16> [[TMP14]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-LATENCY8-NEXT:    [[INTERLEAVED_VEC12:%.*]] = shufflevector <16 x i16> [[TMP22]], <16 x i16> poison, <16 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11, i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
+; CHECK-LATENCY8-NEXT:    store <16 x i16> [[INTERLEAVED_VEC12]], ptr [[TMP18]], align 2
+; CHECK-LATENCY8-NEXT:    [[TMP23:%.*]] = shufflevector <8 x i16> [[TMP11]], <8 x i16> [[TMP15]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-LATENCY8-NEXT:    [[INTERLEAVED_VEC13:%.*]] = shufflevector <16 x i16> [[TMP23]], <16 x i16> poison, <16 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11, i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
+; CHECK-LATENCY8-NEXT:    store <16 x i16> [[INTERLEAVED_VEC13]], ptr [[TMP19]], align 2
+; CHECK-LATENCY8-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-LATENCY8-NEXT:    [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 512
+; CHECK-LATENCY8-NEXT:    br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-LATENCY8:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY8-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY8:       [[EXIT]]:
+; CHECK-LATENCY8-NEXT:    ret void
+;
+; CHECK-A510-LABEL: define void @i16_add2(
+; CHECK-A510-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) #[[ATTR0]] {
+; CHECK-A510-NEXT:  [[ENTRY:.*:]]
+; CHECK-A510-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-A510:       [[VECTOR_PH]]:
+; CHECK-A510-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-A510-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-A510-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-A510:       [[VECTOR_BODY]]:
+; CHECK-A510-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-A510-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 1
+; CHECK-A510-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-A510-NEXT:    [[WIDE_VEC:%.*]] = load <16 x i16>, ptr [[TMP1]], align 2
+; CHECK-A510-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <16 x i16> [[WIDE_VEC]], <16 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
+; CHECK-A510-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <16 x i16> [[WIDE_VEC]], <16 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
+; CHECK-A510-NEXT:    [[TMP2:%.*]] = add <8 x i16> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-A510-NEXT:    [[TMP3:%.*]] = add <8 x i16> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-A510-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[TMP0]]
+; CHECK-A510-NEXT:    [[TMP5:%.*]] = shufflevector <8 x i16> [[TMP2]], <8 x i16> [[TMP3]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-A510-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <16 x i16> [[TMP5]], <16 x i16> poison, <16 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11, i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
+; CHECK-A510-NEXT:    store <16 x i16> [[INTERLEAVED_VEC]], ptr [[TMP4]], align 2
+; CHECK-A510-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-A510-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 512
+; CHECK-A510-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-A510:       [[MIDDLE_BLOCK]]:
+; CHECK-A510-NEXT:    br label %[[EXIT:.*]]
+; CHECK-A510:       [[EXIT]]:
+; CHECK-A510-NEXT:    ret void
+;
+; CHECK-A320-LABEL: define void @i16_add2(
+; CHECK-A320-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) #[[ATTR0]] {
+; CHECK-A320-NEXT:  [[ENTRY:.*:]]
+; CHECK-A320-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-A320:       [[VECTOR_PH]]:
+; CHECK-A320-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-A320-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-A320-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-A320:       [[VECTOR_BODY]]:
+; CHECK-A320-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-A320-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 1
+; CHECK-A320-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-A320-NEXT:    [[WIDE_VEC:%.*]] = load <16 x i16>, ptr [[TMP1]], align 2
+; CHECK-A320-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <16 x i16> [[WIDE_VEC]], <16 x i16> poison, <8 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14>
+; CHECK-A320-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <16 x i16> [[WIDE_VEC]], <16 x i16> poison, <8 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15>
+; CHECK-A320-NEXT:    [[TMP2:%.*]] = add <8 x i16> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP3:%.*]] = add <8 x i16> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[TMP0]]
+; CHECK-A320-NEXT:    [[TMP5:%.*]] = shufflevector <8 x i16> [[TMP2]], <8 x i16> [[TMP3]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-A320-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <16 x i16> [[TMP5]], <16 x i16> poison, <16 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11, i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15>
+; CHECK-A320-NEXT:    store <16 x i16> [[INTERLEAVED_VEC]], ptr [[TMP4]], align 2
+; CHECK-A320-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-A320-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 512
+; CHECK-A320-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-A320:       [[MIDDLE_BLOCK]]:
+; CHECK-A320-NEXT:    br label %[[EXIT:.*]]
+; CHECK-A320:       [[EXIT]]:
+; CHECK-A320-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ %iv.next, %loop ], [ 0, %entry ]
+  %src0 = getelementptr inbounds nuw i16, ptr %src, i64 %iv
+  %src1 = getelementptr inbounds nuw i16, ptr %src0, i64 1
+  %val0 = load i16, ptr %src0, align 2
+  %val1 = load i16, ptr %src1, align 2
+  %add0 = add i16 %val0, %argval
+  %add1 = add i16 %val1, %argval
+  %dst0 = getelementptr inbounds nuw i16, ptr %dst, i64 %iv
+  %dst1 = getelementptr inbounds nuw i16, ptr %dst0, i64 1
+  store i16 %add0, ptr %dst0, align 2
+  store i16 %add1, ptr %dst1, align 2
+  %iv.next = add nuw i64 %iv, 2
+  %cmp = icmp ult i64 %iv.next, 1024
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; FIXME: On a320 and a510 we should be interleaving by 2 here, but
+; latency of interleaved loads is not yet implemented.
+define void @i16_add4(ptr noalias readonly %src, ptr noalias writeonly %dst, i16 %argval) {
+; CHECK-LATENCY1-LABEL: define void @i16_add4(
+; CHECK-LATENCY1-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY1-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY1-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY1:       [[VECTOR_PH]]:
+; CHECK-LATENCY1-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-LATENCY1-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-LATENCY1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY1:       [[VECTOR_BODY]]:
+; CHECK-LATENCY1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY1-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 2
+; CHECK-LATENCY1-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-LATENCY1-NEXT:    [[WIDE_VEC:%.*]] = load <32 x i16>, ptr [[TMP1]], align 2
+; CHECK-LATENCY1-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 0, i32 4, i32 8, i32 12, i32 16, i32 20, i32 24, i32 28>
+; CHECK-LATENCY1-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 1, i32 5, i32 9, i32 13, i32 17, i32 21, i32 25, i32 29>
+; CHECK-LATENCY1-NEXT:    [[STRIDED_VEC2:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 2, i32 6, i32 10, i32 14, i32 18, i32 22, i32 26, i32 30>
+; CHECK-LATENCY1-NEXT:    [[STRIDED_VEC3:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 3, i32 7, i32 11, i32 15, i32 19, i32 23, i32 27, i32 31>
+; CHECK-LATENCY1-NEXT:    [[TMP2:%.*]] = add <8 x i16> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY1-NEXT:    [[TMP3:%.*]] = add <8 x i16> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY1-NEXT:    [[TMP4:%.*]] = add <8 x i16> [[STRIDED_VEC2]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY1-NEXT:    [[TMP5:%.*]] = add <8 x i16> [[STRIDED_VEC3]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY1-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[TMP0]]
+; CHECK-LATENCY1-NEXT:    [[TMP7:%.*]] = shufflevector <8 x i16> [[TMP2]], <8 x i16> [[TMP3]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-LATENCY1-NEXT:    [[TMP8:%.*]] = shufflevector <8 x i16> [[TMP4]], <8 x i16> [[TMP5]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-LATENCY1-NEXT:    [[TMP9:%.*]] = shufflevector <16 x i16> [[TMP7]], <16 x i16> [[TMP8]], <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>
+; CHECK-LATENCY1-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <32 x i16> [[TMP9]], <32 x i16> poison, <32 x i32> <i32 0, i32 8, i32 16, i32 24, i32 1, i32 9, i32 17, i32 25, i32 2, i32 10, i32 18, i32 26, i32 3, i32 11, i32 19, i32 27, i32 4, i32 12, i32 20, i32 28, i32 5, i32 13, i32 21, i32 29, i32 6, i32 14, i32 22, i32 30, i32 7, i32 15, i32 23, i32 31>
+; CHECK-LATENCY1-NEXT:    store <32 x i16> [[INTERLEAVED_VEC]], ptr [[TMP6]], align 2
+; CHECK-LATENCY1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-LATENCY1-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256
+; CHECK-LATENCY1-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-LATENCY1:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY1-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY1:       [[EXIT]]:
+; CHECK-LATENCY1-NEXT:    ret void
+;
+; CHECK-LATENCY2-LABEL: define void @i16_add4(
+; CHECK-LATENCY2-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY2-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY2-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY2:       [[VECTOR_PH]]:
+; CHECK-LATENCY2-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-LATENCY2-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-LATENCY2-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY2:       [[VECTOR_BODY]]:
+; CHECK-LATENCY2-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY2-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 2
+; CHECK-LATENCY2-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-LATENCY2-NEXT:    [[WIDE_VEC:%.*]] = load <32 x i16>, ptr [[TMP1]], align 2
+; CHECK-LATENCY2-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 0, i32 4, i32 8, i32 12, i32 16, i32 20, i32 24, i32 28>
+; CHECK-LATENCY2-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 1, i32 5, i32 9, i32 13, i32 17, i32 21, i32 25, i32 29>
+; CHECK-LATENCY2-NEXT:    [[STRIDED_VEC2:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 2, i32 6, i32 10, i32 14, i32 18, i32 22, i32 26, i32 30>
+; CHECK-LATENCY2-NEXT:    [[STRIDED_VEC3:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 3, i32 7, i32 11, i32 15, i32 19, i32 23, i32 27, i32 31>
+; CHECK-LATENCY2-NEXT:    [[TMP2:%.*]] = add <8 x i16> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY2-NEXT:    [[TMP3:%.*]] = add <8 x i16> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY2-NEXT:    [[TMP4:%.*]] = add <8 x i16> [[STRIDED_VEC2]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY2-NEXT:    [[TMP5:%.*]] = add <8 x i16> [[STRIDED_VEC3]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY2-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[TMP0]]
+; CHECK-LATENCY2-NEXT:    [[TMP7:%.*]] = shufflevector <8 x i16> [[TMP2]], <8 x i16> [[TMP3]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-LATENCY2-NEXT:    [[TMP8:%.*]] = shufflevector <8 x i16> [[TMP4]], <8 x i16> [[TMP5]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-LATENCY2-NEXT:    [[TMP9:%.*]] = shufflevector <16 x i16> [[TMP7]], <16 x i16> [[TMP8]], <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>
+; CHECK-LATENCY2-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <32 x i16> [[TMP9]], <32 x i16> poison, <32 x i32> <i32 0, i32 8, i32 16, i32 24, i32 1, i32 9, i32 17, i32 25, i32 2, i32 10, i32 18, i32 26, i32 3, i32 11, i32 19, i32 27, i32 4, i32 12, i32 20, i32 28, i32 5, i32 13, i32 21, i32 29, i32 6, i32 14, i32 22, i32 30, i32 7, i32 15, i32 23, i32 31>
+; CHECK-LATENCY2-NEXT:    store <32 x i16> [[INTERLEAVED_VEC]], ptr [[TMP6]], align 2
+; CHECK-LATENCY2-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-LATENCY2-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256
+; CHECK-LATENCY2-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-LATENCY2:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY2-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY2:       [[EXIT]]:
+; CHECK-LATENCY2-NEXT:    ret void
+;
+; CHECK-LATENCY8-LABEL: define void @i16_add4(
+; CHECK-LATENCY8-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY8-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY8-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY8:       [[VECTOR_PH]]:
+; CHECK-LATENCY8-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-LATENCY8-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-LATENCY8-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY8:       [[VECTOR_BODY]]:
+; CHECK-LATENCY8-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY8-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 2
+; CHECK-LATENCY8-NEXT:    [[TMP1:%.*]] = add i64 [[TMP0]], 32
+; CHECK-LATENCY8-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-LATENCY8-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[TMP1]]
+; CHECK-LATENCY8-NEXT:    [[WIDE_VEC:%.*]] = load <32 x i16>, ptr [[TMP2]], align 2
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 0, i32 4, i32 8, i32 12, i32 16, i32 20, i32 24, i32 28>
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 1, i32 5, i32 9, i32 13, i32 17, i32 21, i32 25, i32 29>
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC2:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 2, i32 6, i32 10, i32 14, i32 18, i32 22, i32 26, i32 30>
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC3:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 3, i32 7, i32 11, i32 15, i32 19, i32 23, i32 27, i32 31>
+; CHECK-LATENCY8-NEXT:    [[WIDE_VEC4:%.*]] = load <32 x i16>, ptr [[TMP3]], align 2
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC5:%.*]] = shufflevector <32 x i16> [[WIDE_VEC4]], <32 x i16> poison, <8 x i32> <i32 0, i32 4, i32 8, i32 12, i32 16, i32 20, i32 24, i32 28>
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC6:%.*]] = shufflevector <32 x i16> [[WIDE_VEC4]], <32 x i16> poison, <8 x i32> <i32 1, i32 5, i32 9, i32 13, i32 17, i32 21, i32 25, i32 29>
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC7:%.*]] = shufflevector <32 x i16> [[WIDE_VEC4]], <32 x i16> poison, <8 x i32> <i32 2, i32 6, i32 10, i32 14, i32 18, i32 22, i32 26, i32 30>
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC8:%.*]] = shufflevector <32 x i16> [[WIDE_VEC4]], <32 x i16> poison, <8 x i32> <i32 3, i32 7, i32 11, i32 15, i32 19, i32 23, i32 27, i32 31>
+; CHECK-LATENCY8-NEXT:    [[TMP4:%.*]] = add <8 x i16> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP5:%.*]] = add <8 x i16> [[STRIDED_VEC5]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP6:%.*]] = add <8 x i16> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP7:%.*]] = add <8 x i16> [[STRIDED_VEC6]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP8:%.*]] = add <8 x i16> [[STRIDED_VEC2]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP9:%.*]] = add <8 x i16> [[STRIDED_VEC7]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP10:%.*]] = add <8 x i16> [[STRIDED_VEC3]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP11:%.*]] = add <8 x i16> [[STRIDED_VEC8]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP12:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[TMP0]]
+; CHECK-LATENCY8-NEXT:    [[TMP13:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[TMP1]]
+; CHECK-LATENCY8-NEXT:    [[TMP14:%.*]] = shufflevector <8 x i16> [[TMP4]], <8 x i16> [[TMP6]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-LATENCY8-NEXT:    [[TMP15:%.*]] = shufflevector <8 x i16> [[TMP8]], <8 x i16> [[TMP10]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-LATENCY8-NEXT:    [[TMP16:%.*]] = shufflevector <16 x i16> [[TMP14]], <16 x i16> [[TMP15]], <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>
+; CHECK-LATENCY8-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <32 x i16> [[TMP16]], <32 x i16> poison, <32 x i32> <i32 0, i32 8, i32 16, i32 24, i32 1, i32 9, i32 17, i32 25, i32 2, i32 10, i32 18, i32 26, i32 3, i32 11, i32 19, i32 27, i32 4, i32 12, i32 20, i32 28, i32 5, i32 13, i32 21, i32 29, i32 6, i32 14, i32 22, i32 30, i32 7, i32 15, i32 23, i32 31>
+; CHECK-LATENCY8-NEXT:    store <32 x i16> [[INTERLEAVED_VEC]], ptr [[TMP12]], align 2
+; CHECK-LATENCY8-NEXT:    [[TMP17:%.*]] = shufflevector <8 x i16> [[TMP5]], <8 x i16> [[TMP7]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-LATENCY8-NEXT:    [[TMP18:%.*]] = shufflevector <8 x i16> [[TMP9]], <8 x i16> [[TMP11]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-LATENCY8-NEXT:    [[TMP19:%.*]] = shufflevector <16 x i16> [[TMP17]], <16 x i16> [[TMP18]], <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>
+; CHECK-LATENCY8-NEXT:    [[INTERLEAVED_VEC9:%.*]] = shufflevector <32 x i16> [[TMP19]], <32 x i16> poison, <32 x i32> <i32 0, i32 8, i32 16, i32 24, i32 1, i32 9, i32 17, i32 25, i32 2, i32 10, i32 18, i32 26, i32 3, i32 11, i32 19, i32 27, i32 4, i32 12, i32 20, i32 28, i32 5, i32 13, i32 21, i32 29, i32 6, i32 14, i32 22, i32 30, i32 7, i32 15, i32 23, i32 31>
+; CHECK-LATENCY8-NEXT:    store <32 x i16> [[INTERLEAVED_VEC9]], ptr [[TMP13]], align 2
+; CHECK-LATENCY8-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-LATENCY8-NEXT:    [[TMP20:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256
+; CHECK-LATENCY8-NEXT:    br i1 [[TMP20]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-LATENCY8:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY8-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY8:       [[EXIT]]:
+; CHECK-LATENCY8-NEXT:    ret void
+;
+; CHECK-A510-LABEL: define void @i16_add4(
+; CHECK-A510-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) #[[ATTR0]] {
+; CHECK-A510-NEXT:  [[ENTRY:.*:]]
+; CHECK-A510-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-A510:       [[VECTOR_PH]]:
+; CHECK-A510-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-A510-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-A510-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-A510:       [[VECTOR_BODY]]:
+; CHECK-A510-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-A510-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 2
+; CHECK-A510-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-A510-NEXT:    [[WIDE_VEC:%.*]] = load <32 x i16>, ptr [[TMP1]], align 2
+; CHECK-A510-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 0, i32 4, i32 8, i32 12, i32 16, i32 20, i32 24, i32 28>
+; CHECK-A510-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 1, i32 5, i32 9, i32 13, i32 17, i32 21, i32 25, i32 29>
+; CHECK-A510-NEXT:    [[STRIDED_VEC2:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 2, i32 6, i32 10, i32 14, i32 18, i32 22, i32 26, i32 30>
+; CHECK-A510-NEXT:    [[STRIDED_VEC3:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 3, i32 7, i32 11, i32 15, i32 19, i32 23, i32 27, i32 31>
+; CHECK-A510-NEXT:    [[TMP2:%.*]] = add <8 x i16> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-A510-NEXT:    [[TMP3:%.*]] = add <8 x i16> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-A510-NEXT:    [[TMP4:%.*]] = add <8 x i16> [[STRIDED_VEC2]], [[BROADCAST_SPLAT]]
+; CHECK-A510-NEXT:    [[TMP5:%.*]] = add <8 x i16> [[STRIDED_VEC3]], [[BROADCAST_SPLAT]]
+; CHECK-A510-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[TMP0]]
+; CHECK-A510-NEXT:    [[TMP7:%.*]] = shufflevector <8 x i16> [[TMP2]], <8 x i16> [[TMP3]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-A510-NEXT:    [[TMP8:%.*]] = shufflevector <8 x i16> [[TMP4]], <8 x i16> [[TMP5]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-A510-NEXT:    [[TMP9:%.*]] = shufflevector <16 x i16> [[TMP7]], <16 x i16> [[TMP8]], <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>
+; CHECK-A510-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <32 x i16> [[TMP9]], <32 x i16> poison, <32 x i32> <i32 0, i32 8, i32 16, i32 24, i32 1, i32 9, i32 17, i32 25, i32 2, i32 10, i32 18, i32 26, i32 3, i32 11, i32 19, i32 27, i32 4, i32 12, i32 20, i32 28, i32 5, i32 13, i32 21, i32 29, i32 6, i32 14, i32 22, i32 30, i32 7, i32 15, i32 23, i32 31>
+; CHECK-A510-NEXT:    store <32 x i16> [[INTERLEAVED_VEC]], ptr [[TMP6]], align 2
+; CHECK-A510-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-A510-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256
+; CHECK-A510-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-A510:       [[MIDDLE_BLOCK]]:
+; CHECK-A510-NEXT:    br label %[[EXIT:.*]]
+; CHECK-A510:       [[EXIT]]:
+; CHECK-A510-NEXT:    ret void
+;
+; CHECK-A320-LABEL: define void @i16_add4(
+; CHECK-A320-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i16 [[ARGVAL:%.*]]) #[[ATTR0]] {
+; CHECK-A320-NEXT:  [[ENTRY:.*:]]
+; CHECK-A320-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-A320:       [[VECTOR_PH]]:
+; CHECK-A320-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i16> poison, i16 [[ARGVAL]], i64 0
+; CHECK-A320-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i16> [[BROADCAST_SPLATINSERT]], <8 x i16> poison, <8 x i32> zeroinitializer
+; CHECK-A320-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-A320:       [[VECTOR_BODY]]:
+; CHECK-A320-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-A320-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 2
+; CHECK-A320-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-A320-NEXT:    [[WIDE_VEC:%.*]] = load <32 x i16>, ptr [[TMP1]], align 2
+; CHECK-A320-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 0, i32 4, i32 8, i32 12, i32 16, i32 20, i32 24, i32 28>
+; CHECK-A320-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 1, i32 5, i32 9, i32 13, i32 17, i32 21, i32 25, i32 29>
+; CHECK-A320-NEXT:    [[STRIDED_VEC2:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 2, i32 6, i32 10, i32 14, i32 18, i32 22, i32 26, i32 30>
+; CHECK-A320-NEXT:    [[STRIDED_VEC3:%.*]] = shufflevector <32 x i16> [[WIDE_VEC]], <32 x i16> poison, <8 x i32> <i32 3, i32 7, i32 11, i32 15, i32 19, i32 23, i32 27, i32 31>
+; CHECK-A320-NEXT:    [[TMP2:%.*]] = add <8 x i16> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP3:%.*]] = add <8 x i16> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP4:%.*]] = add <8 x i16> [[STRIDED_VEC2]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP5:%.*]] = add <8 x i16> [[STRIDED_VEC3]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i16, ptr [[DST]], i64 [[TMP0]]
+; CHECK-A320-NEXT:    [[TMP7:%.*]] = shufflevector <8 x i16> [[TMP2]], <8 x i16> [[TMP3]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-A320-NEXT:    [[TMP8:%.*]] = shufflevector <8 x i16> [[TMP4]], <8 x i16> [[TMP5]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+; CHECK-A320-NEXT:    [[TMP9:%.*]] = shufflevector <16 x i16> [[TMP7]], <16 x i16> [[TMP8]], <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>
+; CHECK-A320-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <32 x i16> [[TMP9]], <32 x i16> poison, <32 x i32> <i32 0, i32 8, i32 16, i32 24, i32 1, i32 9, i32 17, i32 25, i32 2, i32 10, i32 18, i32 26, i32 3, i32 11, i32 19, i32 27, i32 4, i32 12, i32 20, i32 28, i32 5, i32 13, i32 21, i32 29, i32 6, i32 14, i32 22, i32 30, i32 7, i32 15, i32 23, i32 31>
+; CHECK-A320-NEXT:    store <32 x i16> [[INTERLEAVED_VEC]], ptr [[TMP6]], align 2
+; CHECK-A320-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-A320-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256
+; CHECK-A320-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-A320:       [[MIDDLE_BLOCK]]:
+; CHECK-A320-NEXT:    br label %[[EXIT:.*]]
+; CHECK-A320:       [[EXIT]]:
+; CHECK-A320-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ %iv.next, %loop ], [ 0, %entry ]
+  %src0 = getelementptr inbounds nuw i16, ptr %src, i64 %iv
+  %src1 = getelementptr inbounds nuw i16, ptr %src0, i64 1
+  %src2 = getelementptr inbounds nuw i16, ptr %src0, i64 2
+  %src3 = getelementptr inbounds nuw i16, ptr %src0, i64 3
+  %val0 = load i16, ptr %src0, align 2
+  %val1 = load i16, ptr %src1, align 2
+  %val2 = load i16, ptr %src2, align 2
+  %val3 = load i16, ptr %src3, align 2
+  %add0 = add i16 %val0, %argval
+  %add1 = add i16 %val1, %argval
+  %add2 = add i16 %val2, %argval
+  %add3 = add i16 %val3, %argval
+  %dst0 = getelementptr inbounds nuw i16, ptr %dst, i64 %iv
+  %dst1 = getelementptr inbounds nuw i16, ptr %dst0, i64 1
+  %dst2 = getelementptr inbounds nuw i16, ptr %dst0, i64 2
+  %dst3 = getelementptr inbounds nuw i16, ptr %dst0, i64 3
+  store i16 %add0, ptr %dst0, align 2
+  store i16 %add1, ptr %dst1, align 2
+  store i16 %add2, ptr %dst2, align 2
+  store i16 %add3, ptr %dst3, align 2
+  %iv.next = add nuw i64 %iv, 4
+  %cmp = icmp ult i64 %iv.next, 1024
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+define void @i32_add1(ptr noalias readonly %src, ptr noalias writeonly %dst, i32 %argval) {
+; CHECK-LATENCY1-LABEL: define void @i32_add1(
+; CHECK-LATENCY1-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY1-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY1-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY1:       [[VECTOR_PH]]:
+; CHECK-LATENCY1-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-LATENCY1-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-LATENCY1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY1:       [[VECTOR_BODY]]:
+; CHECK-LATENCY1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY1-NEXT:    [[TMP0:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-LATENCY1-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4
+; CHECK-LATENCY1-NEXT:    [[TMP1:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY1-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[INDEX]]
+; CHECK-LATENCY1-NEXT:    store <4 x i32> [[TMP1]], ptr [[TMP2]], align 4
+; CHECK-LATENCY1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-LATENCY1-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-LATENCY1-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-LATENCY1:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY1-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY1:       [[EXIT]]:
+; CHECK-LATENCY1-NEXT:    ret void
+;
+; CHECK-LATENCY2-LABEL: define void @i32_add1(
+; CHECK-LATENCY2-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY2-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY2-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY2:       [[VECTOR_PH]]:
+; CHECK-LATENCY2-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-LATENCY2-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-LATENCY2-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY2:       [[VECTOR_BODY]]:
+; CHECK-LATENCY2-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY2-NEXT:    [[TMP0:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-LATENCY2-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP0]], i64 4
+; CHECK-LATENCY2-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4
+; CHECK-LATENCY2-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-LATENCY2-NEXT:    [[TMP2:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY2-NEXT:    [[TMP3:%.*]] = add <4 x i32> [[WIDE_LOAD1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY2-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[INDEX]]
+; CHECK-LATENCY2-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP4]], i64 4
+; CHECK-LATENCY2-NEXT:    store <4 x i32> [[TMP2]], ptr [[TMP4]], align 4
+; CHECK-LATENCY2-NEXT:    store <4 x i32> [[TMP3]], ptr [[TMP5]], align 4
+; CHECK-LATENCY2-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-LATENCY2-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-LATENCY2-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-LATENCY2:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY2-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY2:       [[EXIT]]:
+; CHECK-LATENCY2-NEXT:    ret void
+;
+; CHECK-LATENCY8-LABEL: define void @i32_add1(
+; CHECK-LATENCY8-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY8-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY8-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY8:       [[VECTOR_PH]]:
+; CHECK-LATENCY8-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-LATENCY8-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-LATENCY8-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY8:       [[VECTOR_BODY]]:
+; CHECK-LATENCY8-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY8-NEXT:    [[TMP0:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-LATENCY8-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP0]], i64 4
+; CHECK-LATENCY8-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP0]], i64 8
+; CHECK-LATENCY8-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP0]], i64 12
+; CHECK-LATENCY8-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4
+; CHECK-LATENCY8-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-LATENCY8-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-LATENCY8-NEXT:    [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; CHECK-LATENCY8-NEXT:    [[TMP4:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP6:%.*]] = add <4 x i32> [[WIDE_LOAD2]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP7:%.*]] = add <4 x i32> [[WIDE_LOAD3]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP8:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[INDEX]]
+; CHECK-LATENCY8-NEXT:    [[TMP9:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP8]], i64 4
+; CHECK-LATENCY8-NEXT:    [[TMP10:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP8]], i64 8
+; CHECK-LATENCY8-NEXT:    [[TMP11:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP8]], i64 12
+; CHECK-LATENCY8-NEXT:    store <4 x i32> [[TMP4]], ptr [[TMP8]], align 4
+; CHECK-LATENCY8-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP9]], align 4
+; CHECK-LATENCY8-NEXT:    store <4 x i32> [[TMP6]], ptr [[TMP10]], align 4
+; CHECK-LATENCY8-NEXT:    store <4 x i32> [[TMP7]], ptr [[TMP11]], align 4
+; CHECK-LATENCY8-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-LATENCY8-NEXT:    [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-LATENCY8-NEXT:    br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-LATENCY8:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY8-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY8:       [[EXIT]]:
+; CHECK-LATENCY8-NEXT:    ret void
+;
+; CHECK-A510-LABEL: define void @i32_add1(
+; CHECK-A510-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) #[[ATTR0]] {
+; CHECK-A510-NEXT:  [[ENTRY:.*:]]
+; CHECK-A510-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-A510:       [[VECTOR_PH]]:
+; CHECK-A510-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-A510-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-A510-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-A510:       [[VECTOR_BODY]]:
+; CHECK-A510-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-A510-NEXT:    [[TMP0:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-A510-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP0]], i64 4
+; CHECK-A510-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4
+; CHECK-A510-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-A510-NEXT:    [[TMP2:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-A510-NEXT:    [[TMP3:%.*]] = add <4 x i32> [[WIDE_LOAD1]], [[BROADCAST_SPLAT]]
+; CHECK-A510-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[INDEX]]
+; CHECK-A510-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP4]], i64 4
+; CHECK-A510-NEXT:    store <4 x i32> [[TMP2]], ptr [[TMP4]], align 4
+; CHECK-A510-NEXT:    store <4 x i32> [[TMP3]], ptr [[TMP5]], align 4
+; CHECK-A510-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-A510-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-A510-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-A510:       [[MIDDLE_BLOCK]]:
+; CHECK-A510-NEXT:    br label %[[EXIT:.*]]
+; CHECK-A510:       [[EXIT]]:
+; CHECK-A510-NEXT:    ret void
+;
+; CHECK-A320-LABEL: define void @i32_add1(
+; CHECK-A320-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) #[[ATTR0]] {
+; CHECK-A320-NEXT:  [[ENTRY:.*:]]
+; CHECK-A320-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-A320:       [[VECTOR_PH]]:
+; CHECK-A320-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-A320-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-A320-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-A320:       [[VECTOR_BODY]]:
+; CHECK-A320-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-A320-NEXT:    [[TMP0:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-A320-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP0]], i64 4
+; CHECK-A320-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP0]], i64 8
+; CHECK-A320-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP0]], i64 12
+; CHECK-A320-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4
+; CHECK-A320-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-A320-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-A320-NEXT:    [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; CHECK-A320-NEXT:    [[TMP4:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD1]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP6:%.*]] = add <4 x i32> [[WIDE_LOAD2]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP7:%.*]] = add <4 x i32> [[WIDE_LOAD3]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP8:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[INDEX]]
+; CHECK-A320-NEXT:    [[TMP9:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP8]], i64 4
+; CHECK-A320-NEXT:    [[TMP10:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP8]], i64 8
+; CHECK-A320-NEXT:    [[TMP11:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP8]], i64 12
+; CHECK-A320-NEXT:    store <4 x i32> [[TMP4]], ptr [[TMP8]], align 4
+; CHECK-A320-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP9]], align 4
+; CHECK-A320-NEXT:    store <4 x i32> [[TMP6]], ptr [[TMP10]], align 4
+; CHECK-A320-NEXT:    store <4 x i32> [[TMP7]], ptr [[TMP11]], align 4
+; CHECK-A320-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-A320-NEXT:    [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-A320-NEXT:    br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-A320:       [[MIDDLE_BLOCK]]:
+; CHECK-A320-NEXT:    br label %[[EXIT:.*]]
+; CHECK-A320:       [[EXIT]]:
+; CHECK-A320-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ %iv.next, %loop ], [ 0, %entry ]
+  %src0 = getelementptr inbounds nuw i32, ptr %src, i64 %iv
+  %val0 = load i32, ptr %src0, align 4
+  %add0 = add i32 %val0, %argval
+  %dst0 = getelementptr inbounds nuw i32, ptr %dst, i64 %iv
+  store i32 %add0, ptr %dst0, align 4
+  %iv.next = add nuw i64 %iv, 1
+  %cmp = icmp ult i64 %iv.next, 1024
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; FIXME: On a320 and a510 we should be interleaving by 2 here, but
+; latency of interleaved loads is not yet implemented.
+define void @i32_add2(ptr noalias readonly %src, ptr noalias writeonly %dst, i32 %argval) {
+; CHECK-LATENCY1-LABEL: define void @i32_add2(
+; CHECK-LATENCY1-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY1-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY1-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY1:       [[VECTOR_PH]]:
+; CHECK-LATENCY1-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-LATENCY1-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-LATENCY1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY1:       [[VECTOR_BODY]]:
+; CHECK-LATENCY1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY1-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 1
+; CHECK-LATENCY1-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-LATENCY1-NEXT:    [[WIDE_VEC:%.*]] = load <8 x i32>, ptr [[TMP1]], align 4
+; CHECK-LATENCY1-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <8 x i32> [[WIDE_VEC]], <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-LATENCY1-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <8 x i32> [[WIDE_VEC]], <8 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-LATENCY1-NEXT:    [[TMP2:%.*]] = add <4 x i32> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY1-NEXT:    [[TMP3:%.*]] = add <4 x i32> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY1-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP0]]
+; CHECK-LATENCY1-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> [[TMP3]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-LATENCY1-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <8 x i32> [[TMP5]], <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-LATENCY1-NEXT:    store <8 x i32> [[INTERLEAVED_VEC]], ptr [[TMP4]], align 4
+; CHECK-LATENCY1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-LATENCY1-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 512
+; CHECK-LATENCY1-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-LATENCY1:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY1-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY1:       [[EXIT]]:
+; CHECK-LATENCY1-NEXT:    ret void
+;
+; CHECK-LATENCY2-LABEL: define void @i32_add2(
+; CHECK-LATENCY2-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY2-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY2-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY2:       [[VECTOR_PH]]:
+; CHECK-LATENCY2-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-LATENCY2-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-LATENCY2-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY2:       [[VECTOR_BODY]]:
+; CHECK-LATENCY2-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY2-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 1
+; CHECK-LATENCY2-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-LATENCY2-NEXT:    [[WIDE_VEC:%.*]] = load <8 x i32>, ptr [[TMP1]], align 4
+; CHECK-LATENCY2-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <8 x i32> [[WIDE_VEC]], <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-LATENCY2-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <8 x i32> [[WIDE_VEC]], <8 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-LATENCY2-NEXT:    [[TMP2:%.*]] = add <4 x i32> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY2-NEXT:    [[TMP3:%.*]] = add <4 x i32> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY2-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP0]]
+; CHECK-LATENCY2-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> [[TMP3]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-LATENCY2-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <8 x i32> [[TMP5]], <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-LATENCY2-NEXT:    store <8 x i32> [[INTERLEAVED_VEC]], ptr [[TMP4]], align 4
+; CHECK-LATENCY2-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-LATENCY2-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 512
+; CHECK-LATENCY2-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-LATENCY2:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY2-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY2:       [[EXIT]]:
+; CHECK-LATENCY2-NEXT:    ret void
+;
+; CHECK-LATENCY8-LABEL: define void @i32_add2(
+; CHECK-LATENCY8-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY8-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY8-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY8:       [[VECTOR_PH]]:
+; CHECK-LATENCY8-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-LATENCY8-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-LATENCY8-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY8:       [[VECTOR_BODY]]:
+; CHECK-LATENCY8-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY8-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 1
+; CHECK-LATENCY8-NEXT:    [[TMP1:%.*]] = add i64 [[TMP0]], 8
+; CHECK-LATENCY8-NEXT:    [[TMP2:%.*]] = add i64 [[TMP0]], 16
+; CHECK-LATENCY8-NEXT:    [[TMP3:%.*]] = add i64 [[TMP0]], 24
+; CHECK-LATENCY8-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-LATENCY8-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP1]]
+; CHECK-LATENCY8-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP2]]
+; CHECK-LATENCY8-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP3]]
+; CHECK-LATENCY8-NEXT:    [[WIDE_VEC:%.*]] = load <8 x i32>, ptr [[TMP4]], align 4
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <8 x i32> [[WIDE_VEC]], <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <8 x i32> [[WIDE_VEC]], <8 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-LATENCY8-NEXT:    [[WIDE_VEC2:%.*]] = load <8 x i32>, ptr [[TMP5]], align 4
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC3:%.*]] = shufflevector <8 x i32> [[WIDE_VEC2]], <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC4:%.*]] = shufflevector <8 x i32> [[WIDE_VEC2]], <8 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-LATENCY8-NEXT:    [[WIDE_VEC5:%.*]] = load <8 x i32>, ptr [[TMP6]], align 4
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC6:%.*]] = shufflevector <8 x i32> [[WIDE_VEC5]], <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC7:%.*]] = shufflevector <8 x i32> [[WIDE_VEC5]], <8 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-LATENCY8-NEXT:    [[WIDE_VEC8:%.*]] = load <8 x i32>, ptr [[TMP7]], align 4
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC9:%.*]] = shufflevector <8 x i32> [[WIDE_VEC8]], <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-LATENCY8-NEXT:    [[STRIDED_VEC10:%.*]] = shufflevector <8 x i32> [[WIDE_VEC8]], <8 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-LATENCY8-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP9:%.*]] = add <4 x i32> [[STRIDED_VEC3]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP10:%.*]] = add <4 x i32> [[STRIDED_VEC6]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP11:%.*]] = add <4 x i32> [[STRIDED_VEC9]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP12:%.*]] = add <4 x i32> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP13:%.*]] = add <4 x i32> [[STRIDED_VEC4]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP14:%.*]] = add <4 x i32> [[STRIDED_VEC7]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP15:%.*]] = add <4 x i32> [[STRIDED_VEC10]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP16:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP0]]
+; CHECK-LATENCY8-NEXT:    [[TMP17:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP1]]
+; CHECK-LATENCY8-NEXT:    [[TMP18:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP2]]
+; CHECK-LATENCY8-NEXT:    [[TMP19:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP3]]
+; CHECK-LATENCY8-NEXT:    [[TMP20:%.*]] = shufflevector <4 x i32> [[TMP8]], <4 x i32> [[TMP12]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-LATENCY8-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <8 x i32> [[TMP20]], <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-LATENCY8-NEXT:    store <8 x i32> [[INTERLEAVED_VEC]], ptr [[TMP16]], align 4
+; CHECK-LATENCY8-NEXT:    [[TMP21:%.*]] = shufflevector <4 x i32> [[TMP9]], <4 x i32> [[TMP13]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-LATENCY8-NEXT:    [[INTERLEAVED_VEC11:%.*]] = shufflevector <8 x i32> [[TMP21]], <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-LATENCY8-NEXT:    store <8 x i32> [[INTERLEAVED_VEC11]], ptr [[TMP17]], align 4
+; CHECK-LATENCY8-NEXT:    [[TMP22:%.*]] = shufflevector <4 x i32> [[TMP10]], <4 x i32> [[TMP14]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-LATENCY8-NEXT:    [[INTERLEAVED_VEC12:%.*]] = shufflevector <8 x i32> [[TMP22]], <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-LATENCY8-NEXT:    store <8 x i32> [[INTERLEAVED_VEC12]], ptr [[TMP18]], align 4
+; CHECK-LATENCY8-NEXT:    [[TMP23:%.*]] = shufflevector <4 x i32> [[TMP11]], <4 x i32> [[TMP15]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-LATENCY8-NEXT:    [[INTERLEAVED_VEC13:%.*]] = shufflevector <8 x i32> [[TMP23]], <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-LATENCY8-NEXT:    store <8 x i32> [[INTERLEAVED_VEC13]], ptr [[TMP19]], align 4
+; CHECK-LATENCY8-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-LATENCY8-NEXT:    [[TMP24:%.*]] = icmp eq i64 [[INDEX_NEXT]], 512
+; CHECK-LATENCY8-NEXT:    br i1 [[TMP24]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-LATENCY8:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY8-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY8:       [[EXIT]]:
+; CHECK-LATENCY8-NEXT:    ret void
+;
+; CHECK-A510-LABEL: define void @i32_add2(
+; CHECK-A510-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) #[[ATTR0]] {
+; CHECK-A510-NEXT:  [[ENTRY:.*:]]
+; CHECK-A510-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-A510:       [[VECTOR_PH]]:
+; CHECK-A510-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-A510-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-A510-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-A510:       [[VECTOR_BODY]]:
+; CHECK-A510-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-A510-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 1
+; CHECK-A510-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-A510-NEXT:    [[WIDE_VEC:%.*]] = load <8 x i32>, ptr [[TMP1]], align 4
+; CHECK-A510-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <8 x i32> [[WIDE_VEC]], <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-A510-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <8 x i32> [[WIDE_VEC]], <8 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-A510-NEXT:    [[TMP2:%.*]] = add <4 x i32> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-A510-NEXT:    [[TMP3:%.*]] = add <4 x i32> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-A510-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP0]]
+; CHECK-A510-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> [[TMP3]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-A510-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <8 x i32> [[TMP5]], <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-A510-NEXT:    store <8 x i32> [[INTERLEAVED_VEC]], ptr [[TMP4]], align 4
+; CHECK-A510-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-A510-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 512
+; CHECK-A510-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-A510:       [[MIDDLE_BLOCK]]:
+; CHECK-A510-NEXT:    br label %[[EXIT:.*]]
+; CHECK-A510:       [[EXIT]]:
+; CHECK-A510-NEXT:    ret void
+;
+; CHECK-A320-LABEL: define void @i32_add2(
+; CHECK-A320-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) #[[ATTR0]] {
+; CHECK-A320-NEXT:  [[ENTRY:.*:]]
+; CHECK-A320-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-A320:       [[VECTOR_PH]]:
+; CHECK-A320-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-A320-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-A320-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-A320:       [[VECTOR_BODY]]:
+; CHECK-A320-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-A320-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 1
+; CHECK-A320-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-A320-NEXT:    [[WIDE_VEC:%.*]] = load <8 x i32>, ptr [[TMP1]], align 4
+; CHECK-A320-NEXT:    [[STRIDED_VEC:%.*]] = shufflevector <8 x i32> [[WIDE_VEC]], <8 x i32> poison, <4 x i32> <i32 0, i32 2, i32 4, i32 6>
+; CHECK-A320-NEXT:    [[STRIDED_VEC1:%.*]] = shufflevector <8 x i32> [[WIDE_VEC]], <8 x i32> poison, <4 x i32> <i32 1, i32 3, i32 5, i32 7>
+; CHECK-A320-NEXT:    [[TMP2:%.*]] = add <4 x i32> [[STRIDED_VEC]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP3:%.*]] = add <4 x i32> [[STRIDED_VEC1]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP0]]
+; CHECK-A320-NEXT:    [[TMP5:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> [[TMP3]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+; CHECK-A320-NEXT:    [[INTERLEAVED_VEC:%.*]] = shufflevector <8 x i32> [[TMP5]], <8 x i32> poison, <8 x i32> <i32 0, i32 4, i32 1, i32 5, i32 2, i32 6, i32 3, i32 7>
+; CHECK-A320-NEXT:    store <8 x i32> [[INTERLEAVED_VEC]], ptr [[TMP4]], align 4
+; CHECK-A320-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-A320-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], 512
+; CHECK-A320-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-A320:       [[MIDDLE_BLOCK]]:
+; CHECK-A320-NEXT:    br label %[[EXIT:.*]]
+; CHECK-A320:       [[EXIT]]:
+; CHECK-A320-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ %iv.next, %loop ], [ 0, %entry ]
+  %src0 = getelementptr inbounds nuw i32, ptr %src, i64 %iv
+  %src1 = getelementptr inbounds nuw i32, ptr %src0, i64 1
+  %val0 = load i32, ptr %src0, align 4
+  %val1 = load i32, ptr %src1, align 4
+  %add0 = add i32 %val0, %argval
+  %add1 = add i32 %val1, %argval
+  %dst0 = getelementptr inbounds nuw i32, ptr %dst, i64 %iv
+  %dst1 = getelementptr inbounds nuw i32, ptr %dst0, i64 1
+  store i32 %add0, ptr %dst0, align 4
+  store i32 %add1, ptr %dst1, align 4
+  %iv.next = add nuw i64 %iv, 2
+  %cmp = icmp ult i64 %iv.next, 1024
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+define void @i32_add4(ptr noalias readonly %src, ptr noalias writeonly %dst, i32 %argval) {
+; CHECK-LATENCY1-LABEL: define void @i32_add4(
+; CHECK-LATENCY1-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY1-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY1-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY1:       [[VECTOR_PH]]:
+; CHECK-LATENCY1-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-LATENCY1-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-LATENCY1-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY1:       [[VECTOR_BODY]]:
+; CHECK-LATENCY1-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY1-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 2
+; CHECK-LATENCY1-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-LATENCY1-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-LATENCY1-NEXT:    [[TMP2:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY1-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP0]]
+; CHECK-LATENCY1-NEXT:    store <4 x i32> [[TMP2]], ptr [[TMP3]], align 4
+; CHECK-LATENCY1-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 1
+; CHECK-LATENCY1-NEXT:    [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256
+; CHECK-LATENCY1-NEXT:    br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-LATENCY1:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY1-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY1:       [[EXIT]]:
+; CHECK-LATENCY1-NEXT:    ret void
+;
+; CHECK-LATENCY2-LABEL: define void @i32_add4(
+; CHECK-LATENCY2-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY2-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY2-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY2:       [[VECTOR_PH]]:
+; CHECK-LATENCY2-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-LATENCY2-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-LATENCY2-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY2:       [[VECTOR_BODY]]:
+; CHECK-LATENCY2-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY2-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 2
+; CHECK-LATENCY2-NEXT:    [[TMP1:%.*]] = add i64 [[TMP0]], 4
+; CHECK-LATENCY2-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-LATENCY2-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP1]]
+; CHECK-LATENCY2-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-LATENCY2-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; CHECK-LATENCY2-NEXT:    [[TMP4:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY2-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY2-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP0]]
+; CHECK-LATENCY2-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP1]]
+; CHECK-LATENCY2-NEXT:    store <4 x i32> [[TMP4]], ptr [[TMP6]], align 4
+; CHECK-LATENCY2-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP7]], align 4
+; CHECK-LATENCY2-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 2
+; CHECK-LATENCY2-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256
+; CHECK-LATENCY2-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-LATENCY2:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY2-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY2:       [[EXIT]]:
+; CHECK-LATENCY2-NEXT:    ret void
+;
+; CHECK-LATENCY8-LABEL: define void @i32_add4(
+; CHECK-LATENCY8-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) {
+; CHECK-LATENCY8-NEXT:  [[ENTRY:.*:]]
+; CHECK-LATENCY8-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-LATENCY8:       [[VECTOR_PH]]:
+; CHECK-LATENCY8-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-LATENCY8-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-LATENCY8-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-LATENCY8:       [[VECTOR_BODY]]:
+; CHECK-LATENCY8-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-LATENCY8-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 2
+; CHECK-LATENCY8-NEXT:    [[TMP1:%.*]] = add i64 [[TMP0]], 4
+; CHECK-LATENCY8-NEXT:    [[TMP2:%.*]] = add i64 [[TMP0]], 8
+; CHECK-LATENCY8-NEXT:    [[TMP3:%.*]] = add i64 [[TMP0]], 12
+; CHECK-LATENCY8-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-LATENCY8-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP1]]
+; CHECK-LATENCY8-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP2]]
+; CHECK-LATENCY8-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP3]]
+; CHECK-LATENCY8-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-LATENCY8-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4
+; CHECK-LATENCY8-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP6]], align 4
+; CHECK-LATENCY8-NEXT:    [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4
+; CHECK-LATENCY8-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP9:%.*]] = add <4 x i32> [[WIDE_LOAD1]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP10:%.*]] = add <4 x i32> [[WIDE_LOAD2]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP11:%.*]] = add <4 x i32> [[WIDE_LOAD3]], [[BROADCAST_SPLAT]]
+; CHECK-LATENCY8-NEXT:    [[TMP12:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP0]]
+; CHECK-LATENCY8-NEXT:    [[TMP13:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP1]]
+; CHECK-LATENCY8-NEXT:    [[TMP14:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP2]]
+; CHECK-LATENCY8-NEXT:    [[TMP15:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP3]]
+; CHECK-LATENCY8-NEXT:    store <4 x i32> [[TMP8]], ptr [[TMP12]], align 4
+; CHECK-LATENCY8-NEXT:    store <4 x i32> [[TMP9]], ptr [[TMP13]], align 4
+; CHECK-LATENCY8-NEXT:    store <4 x i32> [[TMP10]], ptr [[TMP14]], align 4
+; CHECK-LATENCY8-NEXT:    store <4 x i32> [[TMP11]], ptr [[TMP15]], align 4
+; CHECK-LATENCY8-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-LATENCY8-NEXT:    [[TMP16:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256
+; CHECK-LATENCY8-NEXT:    br i1 [[TMP16]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-LATENCY8:       [[MIDDLE_BLOCK]]:
+; CHECK-LATENCY8-NEXT:    br label %[[EXIT:.*]]
+; CHECK-LATENCY8:       [[EXIT]]:
+; CHECK-LATENCY8-NEXT:    ret void
+;
+; CHECK-A510-LABEL: define void @i32_add4(
+; CHECK-A510-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) #[[ATTR0]] {
+; CHECK-A510-NEXT:  [[ENTRY:.*:]]
+; CHECK-A510-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-A510:       [[VECTOR_PH]]:
+; CHECK-A510-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-A510-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-A510-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-A510:       [[VECTOR_BODY]]:
+; CHECK-A510-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-A510-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 2
+; CHECK-A510-NEXT:    [[TMP1:%.*]] = add i64 [[TMP0]], 4
+; CHECK-A510-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-A510-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP1]]
+; CHECK-A510-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-A510-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; CHECK-A510-NEXT:    [[TMP4:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-A510-NEXT:    [[TMP5:%.*]] = add <4 x i32> [[WIDE_LOAD1]], [[BROADCAST_SPLAT]]
+; CHECK-A510-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP0]]
+; CHECK-A510-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP1]]
+; CHECK-A510-NEXT:    store <4 x i32> [[TMP4]], ptr [[TMP6]], align 4
+; CHECK-A510-NEXT:    store <4 x i32> [[TMP5]], ptr [[TMP7]], align 4
+; CHECK-A510-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 2
+; CHECK-A510-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256
+; CHECK-A510-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-A510:       [[MIDDLE_BLOCK]]:
+; CHECK-A510-NEXT:    br label %[[EXIT:.*]]
+; CHECK-A510:       [[EXIT]]:
+; CHECK-A510-NEXT:    ret void
+;
+; CHECK-A320-LABEL: define void @i32_add4(
+; CHECK-A320-SAME: ptr noalias readonly [[SRC:%.*]], ptr noalias writeonly [[DST:%.*]], i32 [[ARGVAL:%.*]]) #[[ATTR0]] {
+; CHECK-A320-NEXT:  [[ENTRY:.*:]]
+; CHECK-A320-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK-A320:       [[VECTOR_PH]]:
+; CHECK-A320-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[ARGVAL]], i64 0
+; CHECK-A320-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-A320-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-A320:       [[VECTOR_BODY]]:
+; CHECK-A320-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-A320-NEXT:    [[TMP0:%.*]] = shl i64 [[INDEX]], 2
+; CHECK-A320-NEXT:    [[TMP1:%.*]] = add i64 [[TMP0]], 4
+; CHECK-A320-NEXT:    [[TMP2:%.*]] = add i64 [[TMP0]], 8
+; CHECK-A320-NEXT:    [[TMP3:%.*]] = add i64 [[TMP0]], 12
+; CHECK-A320-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP0]]
+; CHECK-A320-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP1]]
+; CHECK-A320-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP2]]
+; CHECK-A320-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[TMP3]]
+; CHECK-A320-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-A320-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP5]], align 4
+; CHECK-A320-NEXT:    [[WIDE_LOAD2:%.*]] = load <4 x i32>, ptr [[TMP6]], align 4
+; CHECK-A320-NEXT:    [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4
+; CHECK-A320-NEXT:    [[TMP8:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP9:%.*]] = add <4 x i32> [[WIDE_LOAD1]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP10:%.*]] = add <4 x i32> [[WIDE_LOAD2]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP11:%.*]] = add <4 x i32> [[WIDE_LOAD3]], [[BROADCAST_SPLAT]]
+; CHECK-A320-NEXT:    [[TMP12:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP0]]
+; CHECK-A320-NEXT:    [[TMP13:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP1]]
+; CHECK-A320-NEXT:    [[TMP14:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP2]]
+; CHECK-A320-NEXT:    [[TMP15:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[TMP3]]
+; CHECK-A320-NEXT:    store <4 x i32> [[TMP8]], ptr [[TMP12]], align 4
+; CHECK-A320-NEXT:    store <4 x i32> [[TMP9]], ptr [[TMP13]], align 4
+; CHECK-A320-NEXT:    store <4 x i32> [[TMP10]], ptr [[TMP14]], align 4
+; CHECK-A320-NEXT:    store <4 x i32> [[TMP11]], ptr [[TMP15]], align 4
+; CHECK-A320-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-A320-NEXT:    [[TMP16:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256
+; CHECK-A320-NEXT:    br i1 [[TMP16]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-A320:       [[MIDDLE_BLOCK]]:
+; CHECK-A320-NEXT:    br label %[[EXIT:.*]]
+; CHECK-A320:       [[EXIT]]:
+; CHECK-A320-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ %iv.next, %loop ], [ 0, %entry ]
+  %src0 = getelementptr inbounds nuw i32, ptr %src, i64 %iv
+  %src1 = getelementptr inbounds nuw i32, ptr %src0, i64 1
+  %src2 = getelementptr inbounds nuw i32, ptr %src0, i64 2
+  %src3 = getelementptr inbounds nuw i32, ptr %src0, i64 3
+  %val0 = load i32, ptr %src0, align 4
+  %val1 = load i32, ptr %src1, align 4
+  %val2 = load i32, ptr %src2, align 4
+  %val3 = load i32, ptr %src3, align 4
+  %add0 = add i32 %val0, %argval
+  %add1 = add i32 %val1, %argval
+  %add2 = add i32 %val2, %argval
+  %add3 = add i32 %val3, %argval
+  %dst0 = getelementptr inbounds nuw i32, ptr %dst, i64 %iv
+  %dst1 = getelementptr inbounds nuw i32, ptr %dst0, i64 1
+  %dst2 = getelementptr inbounds nuw i32, ptr %dst0, i64 2
+  %dst3 = getelementptr inbounds nuw i32, ptr %dst0, i64 3
+  store i32 %add0, ptr %dst0, align 4
+  store i32 %add1, ptr %dst1, align 4
+  store i32 %add2, ptr %dst2, align 4
+  store i32 %add3, ptr %dst3, align 4
+  %iv.next = add nuw i64 %iv, 4
+  %cmp = icmp ult i64 %iv.next, 1024
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}



More information about the llvm-commits mailing list