[llvm] [SimplifyCFG] Avoid sinking loads/stores that impact vectorization (PR #222587)

via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 10 03:23:47 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Ashutosh Nema (nema-ashutosh)

<details>
<summary>Changes</summary>

Sinking load/store memory ops through a pointer PHI forces later vectorization to use suboptimal code generation. Keep them separate when independent masked loads or stores are cheaper, matching the existing store-sinking profitability guard and applying it to loads as well.

Fixes #<!-- -->222516

---

Patch is 35.92 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/222587.diff


5 Files Affected:

- (modified) llvm/lib/Transforms/Utils/SimplifyCFG.cpp (+118-3) 
- (added) llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-different-pointers.ll (+148) 
- (added) llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-gather-cost.ll (+189) 
- (added) llvm/test/Transforms/SimplifyCFG/X86/sink-common-store-different-pointers.ll (+152) 
- (added) llvm/test/Transforms/SimplifyCFG/X86/sink-common-store-scatter-cost.ll (+190) 


``````````diff
diff --git a/llvm/lib/Transforms/Utils/SimplifyCFG.cpp b/llvm/lib/Transforms/Utils/SimplifyCFG.cpp
index 6a2601487b178..9eb0d7e4bd5ee 100644
--- a/llvm/lib/Transforms/Utils/SimplifyCFG.cpp
+++ b/llvm/lib/Transforms/Utils/SimplifyCFG.cpp
@@ -2420,10 +2420,106 @@ static void sinkLastInstruction(ArrayRef<BasicBlock*> Blocks) {
   }
 }
 
+/// Estimate whether commoning \p NumMemOps loads or stores like \p I, which
+/// access distinct addresses, penalizes a later vectorizer. Commoning needs a
+/// pointer PHI, so the result can only be widened as a gather or scatter, while
+/// separate blocks allow one independent masked load or store each.
+static bool sinkingMemOpsPenalizesVectorization(const TargetTransformInfo &TTI,
+                                                Instruction *I,
+                                                unsigned NumMemOps) {
+  bool IsLoad = isa<LoadInst>(I);
+  assert((IsLoad || isa<StoreInst>(I)) && "Expected a load or store");
+  Type *ScalarTy = getLoadStoreType(I);
+  Align Alignment = getLoadStoreAlignment(I);
+  unsigned AS = getLoadStoreAddressSpace(I);
+  Intrinsic::ID MaskedID =
+      IsLoad ? Intrinsic::masked_load : Intrinsic::masked_store;
+  Intrinsic::ID GatherScatterID =
+      IsLoad ? Intrinsic::masked_gather : Intrinsic::masked_scatter;
+
+  // There is nothing to preserve unless the operations can be widened in place.
+  if (!VectorType::isValidElementType(ScalarTy) ||
+      !(IsLoad ? TTI.isLegalMaskedLoad(ScalarTy, Alignment, AS)
+               : TTI.isLegalMaskedStore(ScalarTy, Alignment, AS)))
+    return false;
+
+  // Without a gather or scatter the commoned operation cannot be widened.
+  if (!(IsLoad ? TTI.isLegalMaskedGather(ScalarTy, Alignment)
+               : TTI.isLegalMaskedScatter(ScalarTy, Alignment)))
+    return true;
+
+  TypeSize EltWidth = I->getDataLayout().getTypeSizeInBits(ScalarTy);
+  if (EltWidth.isScalable() || EltWidth.isZero())
+    return true;
+
+  auto PenalizesForVF = [&](ElementCount VF) {
+    auto *VecTy = VectorType::get(ScalarTy, VF);
+    Value *Ptr = getLoadStorePointerOperand(I);
+    auto *PtrVecTy = VectorType::get(Ptr->getType(), VF);
+    constexpr TargetTransformInfo::TargetCostKind CostKind =
+        TargetTransformInfo::TCK_RecipThroughput;
+
+    InstructionCost GatherScatterCost =
+        TTI.getAddressComputationCost(PtrVecTy, nullptr, nullptr, CostKind) +
+        TTI.getMemIntrinsicInstrCost(
+            MemIntrinsicCostAttributes(GatherScatterID, VecTy, Ptr,
+                                       /*VariableMask=*/true, Alignment, I),
+            CostKind);
+    InstructionCost MaskedCost =
+        TTI.getMemIntrinsicInstrCost(
+            MemIntrinsicCostAttributes(MaskedID, VecTy, Alignment, AS),
+            CostKind) *
+        NumMemOps;
+
+    if (!GatherScatterCost.isValid())
+      return true;
+    if (!MaskedCost.isValid())
+      return false;
+
+    LLVM_DEBUG(dbgs() << "SINK: " << (VF.isScalable() ? "scalable" : "fixed")
+                      << " VF " << VF.getKnownMinValue() << ": "
+                      << (IsLoad ? "gather" : "scatter") << " cost "
+                      << GatherScatterCost << " vs " << NumMemOps << " masked "
+                      << (IsLoad ? "loads " : "stores ") << MaskedCost << "\n");
+    return GatherScatterCost >= MaskedCost;
+  };
+
+  // One vector register's worth of elements, at vscale=1 for scalable vectors.
+  auto VFFrom = [&](TargetTransformInfo::RegisterKind RK,
+                    bool Scalable) -> std::optional<ElementCount> {
+    TypeSize RegWidth = TTI.getRegisterBitWidth(RK);
+    if (RegWidth.isScalable() != Scalable || RegWidth.isZero())
+      return std::nullopt;
+    unsigned NumElts = RegWidth.getKnownMinValue() / EltWidth.getFixedValue();
+    if (NumElts < (Scalable ? 1u : 2u))
+      return std::nullopt;
+    return ElementCount::get(NumElts, Scalable);
+  };
+
+  std::optional<ElementCount> FixedVF =
+      VFFrom(TargetTransformInfo::RGK_FixedWidthVector, /*Scalable=*/false);
+  std::optional<ElementCount> ScalableVF =
+      VFFrom(TargetTransformInfo::RGK_ScalableVector, /*Scalable=*/true);
+
+  // Both forms are legal but no vector factor can be formed, so the target
+  // vectorizes in a way this cannot reason about. Keep the operations separate.
+  if (!FixedVF && !ScalableVF)
+    return true;
+
+  // A vectorizer may pick either kind of vector, so keep the operations
+  // separate if commoning them does not pay off for one of them.
+  bool Penalizes = false;
+  if (FixedVF)
+    Penalizes |= PenalizesForVF(*FixedVF);
+  if (ScalableVF)
+    Penalizes |= PenalizesForVF(*ScalableVF);
+  return Penalizes;
+}
+
 /// Check whether BB's predecessors end with unconditional branches. If it is
 /// true, sink any common code from the predecessors to BB.
-static bool sinkCommonCodeFromPredecessors(BasicBlock *BB,
-                                           DomTreeUpdater *DTU) {
+static bool sinkCommonCodeFromPredecessors(BasicBlock *BB, DomTreeUpdater *DTU,
+                                           const TargetTransformInfo &TTI) {
   // We support two situations:
   //   (1) all incoming arcs are unconditional
   //   (2) there are non-unconditional incoming arcs
@@ -2523,10 +2619,29 @@ static bool sinkCommonCodeFromPredecessors(BasicBlock *BB,
       return false;
     };
 
+    // Check whether the memory operations in \p Insts all access the same
+    // address, so that commoning them needs no PHI for the pointer operand.
+    auto HaveSameMemAddress = [](ArrayRef<Instruction *> Insts) {
+      Value *Ptr = getLoadStorePointerOperand(Insts.front());
+      auto *PtrI = dyn_cast<Instruction>(Ptr);
+      return all_of(drop_begin(Insts), [&](Instruction *I) {
+        Value *OtherPtr = getLoadStorePointerOperand(I);
+        if (OtherPtr == Ptr)
+          return true;
+        auto *OtherPtrI = dyn_cast<Instruction>(OtherPtr);
+        return PtrI && OtherPtrI && PtrI->isIdenticalTo(OtherPtrI);
+      });
+    };
+
     // Okay, we *could* sink last ScanIdx instructions. But how many can we
     // actually sink before encountering instruction that is unprofitable to
     // sink?
     auto ProfitableToSinkInstruction = [&](LockstepReverseIterator<true> &LRI) {
+      ArrayRef<Instruction *> Insts = *LRI;
+      if (isa<LoadInst, StoreInst>(Insts[0]) && !HaveSameMemAddress(Insts) &&
+          sinkingMemOpsPenalizesVectorization(TTI, Insts[0], Insts.size()))
+        return false;
+
       unsigned NumPHIInsts = 0;
       for (Use &U : (*LRI)[0]->operands()) {
         auto It = PHIOperands.find(&U);
@@ -9319,7 +9434,7 @@ bool SimplifyCFGOpt::simplifyOnce(BasicBlock *BB) {
     return true;
 
   if (SinkCommon && Options.SinkCommonInsts) {
-    if (sinkCommonCodeFromPredecessors(BB, DTU) ||
+    if (sinkCommonCodeFromPredecessors(BB, DTU, TTI) ||
         mergeCompatibleInvokes(BB, DTU)) {
       // sinkCommonCodeFromPredecessors() does not automatically CSE PHI's,
       // so we may now how duplicate PHI's.
diff --git a/llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-different-pointers.ll b/llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-different-pointers.ll
new file mode 100644
index 0000000000000..5e2eb7f64c0af
--- /dev/null
+++ b/llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-different-pointers.ll
@@ -0,0 +1,148 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt < %s -mtriple=x86_64-- -mattr=+avx512f -passes='simplifycfg<sink-common-insts>' -S | FileCheck %s --check-prefixes=MASKED
+; RUN: opt < %s -mtriple=x86_64-- -mattr=+avx2 -passes='simplifycfg<sink-common-insts>' -S | FileCheck %s --check-prefixes=NOGATHER
+; RUN: opt < %s -mtriple=x86_64-- -mattr=+sse2 -passes='simplifycfg<sink-common-insts>' -S | FileCheck %s --check-prefixes=SCALAR
+
+; Loads from different address expressions can only be commoned through a
+; pointer PHI. On a target with masked loads, keep them in their original
+; blocks so that a vectorizer can widen each one on its own. AVX2 reaches the
+; same result without a gather, which cannot be widened at all.
+define i32 @different_load_pointers(i1 %cond, ptr %a, ptr %b, i64 %index) {
+; MASKED-LABEL: define i32 @different_load_pointers(
+; MASKED-SAME: i1 [[COND:%.*]], ptr [[A:%.*]], ptr [[B:%.*]], i64 [[INDEX:%.*]]) #[[ATTR0:[0-9]+]] {
+; MASKED-NEXT:  [[ENTRY:.*:]]
+; MASKED-NEXT:    br i1 [[COND]], label %[[IF_THEN:.*]], label %[[IF_ELSE:.*]]
+; MASKED:       [[IF_THEN]]:
+; MASKED-NEXT:    [[A_GEP:%.*]] = getelementptr i32, ptr [[A]], i64 [[INDEX]]
+; MASKED-NEXT:    [[X:%.*]] = load i32, ptr [[A_GEP]], align 4
+; MASKED-NEXT:    br label %[[EXIT:.*]]
+; MASKED:       [[IF_ELSE]]:
+; MASKED-NEXT:    [[B_GEP:%.*]] = getelementptr i32, ptr [[B]], i64 [[INDEX]]
+; MASKED-NEXT:    [[Y:%.*]] = load i32, ptr [[B_GEP]], align 4
+; MASKED-NEXT:    br label %[[EXIT]]
+; MASKED:       [[EXIT]]:
+; MASKED-NEXT:    [[R:%.*]] = phi i32 [ [[X]], %[[IF_THEN]] ], [ [[Y]], %[[IF_ELSE]] ]
+; MASKED-NEXT:    ret i32 [[R]]
+;
+; NOGATHER-LABEL: define i32 @different_load_pointers(
+; NOGATHER-SAME: i1 [[COND:%.*]], ptr [[A:%.*]], ptr [[B:%.*]], i64 [[INDEX:%.*]]) #[[ATTR0:[0-9]+]] {
+; NOGATHER-NEXT:  [[ENTRY:.*:]]
+; NOGATHER-NEXT:    br i1 [[COND]], label %[[IF_THEN:.*]], label %[[IF_ELSE:.*]]
+; NOGATHER:       [[IF_THEN]]:
+; NOGATHER-NEXT:    [[A_GEP:%.*]] = getelementptr i32, ptr [[A]], i64 [[INDEX]]
+; NOGATHER-NEXT:    [[X:%.*]] = load i32, ptr [[A_GEP]], align 4
+; NOGATHER-NEXT:    br label %[[EXIT:.*]]
+; NOGATHER:       [[IF_ELSE]]:
+; NOGATHER-NEXT:    [[B_GEP:%.*]] = getelementptr i32, ptr [[B]], i64 [[INDEX]]
+; NOGATHER-NEXT:    [[Y:%.*]] = load i32, ptr [[B_GEP]], align 4
+; NOGATHER-NEXT:    br label %[[EXIT]]
+; NOGATHER:       [[EXIT]]:
+; NOGATHER-NEXT:    [[R:%.*]] = phi i32 [ [[X]], %[[IF_THEN]] ], [ [[Y]], %[[IF_ELSE]] ]
+; NOGATHER-NEXT:    ret i32 [[R]]
+;
+; SCALAR-LABEL: define i32 @different_load_pointers(
+; SCALAR-SAME: i1 [[COND:%.*]], ptr [[A:%.*]], ptr [[B:%.*]], i64 [[INDEX:%.*]]) #[[ATTR0:[0-9]+]] {
+; SCALAR-NEXT:  [[ENTRY:.*:]]
+; SCALAR-NEXT:    [[A_B:%.*]] = select i1 [[COND]], ptr [[A]], ptr [[B]]
+; SCALAR-NEXT:    [[B_GEP:%.*]] = getelementptr i32, ptr [[A_B]], i64 [[INDEX]]
+; SCALAR-NEXT:    [[Y:%.*]] = load i32, ptr [[B_GEP]], align 4
+; SCALAR-NEXT:    ret i32 [[Y]]
+;
+entry:
+  br i1 %cond, label %if.then, label %if.else
+
+if.then:
+  %a.gep = getelementptr i32, ptr %a, i64 %index
+  %x = load i32, ptr %a.gep, align 4
+  br label %exit
+
+if.else:
+  %b.gep = getelementptr i32, ptr %b, i64 %index
+  %y = load i32, ptr %b.gep, align 4
+  br label %exit
+
+exit:
+  %r = phi i32 [ %x, %if.then ], [ %y, %if.else ]
+  ret i32 %r
+}
+
+; A pointer PHI is not needed when both loads use the same address, so these
+; are commoned on any target.
+define i32 @same_load_pointer(i1 %cond, ptr %a) {
+; MASKED-LABEL: define i32 @same_load_pointer(
+; MASKED-SAME: i1 [[COND:%.*]], ptr [[A:%.*]]) #[[ATTR0]] {
+; MASKED-NEXT:  [[ENTRY:.*:]]
+; MASKED-NEXT:    [[X:%.*]] = load i32, ptr [[A]], align 4
+; MASKED-NEXT:    ret i32 [[X]]
+;
+; NOGATHER-LABEL: define i32 @same_load_pointer(
+; NOGATHER-SAME: i1 [[COND:%.*]], ptr [[A:%.*]]) #[[ATTR0]] {
+; NOGATHER-NEXT:  [[ENTRY:.*:]]
+; NOGATHER-NEXT:    [[X:%.*]] = load i32, ptr [[A]], align 4
+; NOGATHER-NEXT:    ret i32 [[X]]
+;
+; SCALAR-LABEL: define i32 @same_load_pointer(
+; SCALAR-SAME: i1 [[COND:%.*]], ptr [[A:%.*]]) #[[ATTR0]] {
+; SCALAR-NEXT:  [[ENTRY:.*:]]
+; SCALAR-NEXT:    [[X:%.*]] = load i32, ptr [[A]], align 4
+; SCALAR-NEXT:    ret i32 [[X]]
+;
+entry:
+  br i1 %cond, label %if.then, label %if.else
+
+if.then:
+  %x = load i32, ptr %a, align 4
+  br label %exit
+
+if.else:
+  %y = load i32, ptr %a, align 4
+  br label %exit
+
+exit:
+  %r = phi i32 [ %x, %if.then ], [ %y, %if.else ]
+  ret i32 %r
+}
+
+; Types that the target cannot load under a mask are still commoned.
+define i8 @different_load_pointers_unsupported_type(i1 %cond, ptr %a, ptr %b, i64 %index) {
+; MASKED-LABEL: define i8 @different_load_pointers_unsupported_type(
+; MASKED-SAME: i1 [[COND:%.*]], ptr [[A:%.*]], ptr [[B:%.*]], i64 [[INDEX:%.*]]) #[[ATTR0]] {
+; MASKED-NEXT:  [[ENTRY:.*:]]
+; MASKED-NEXT:    [[A_B:%.*]] = select i1 [[COND]], ptr [[A]], ptr [[B]]
+; MASKED-NEXT:    [[B_GEP:%.*]] = getelementptr i8, ptr [[A_B]], i64 [[INDEX]]
+; MASKED-NEXT:    [[Y:%.*]] = load i8, ptr [[B_GEP]], align 1
+; MASKED-NEXT:    ret i8 [[Y]]
+;
+; NOGATHER-LABEL: define i8 @different_load_pointers_unsupported_type(
+; NOGATHER-SAME: i1 [[COND:%.*]], ptr [[A:%.*]], ptr [[B:%.*]], i64 [[INDEX:%.*]]) #[[ATTR0]] {
+; NOGATHER-NEXT:  [[ENTRY:.*:]]
+; NOGATHER-NEXT:    [[A_B:%.*]] = select i1 [[COND]], ptr [[A]], ptr [[B]]
+; NOGATHER-NEXT:    [[B_GEP:%.*]] = getelementptr i8, ptr [[A_B]], i64 [[INDEX]]
+; NOGATHER-NEXT:    [[Y:%.*]] = load i8, ptr [[B_GEP]], align 1
+; NOGATHER-NEXT:    ret i8 [[Y]]
+;
+; SCALAR-LABEL: define i8 @different_load_pointers_unsupported_type(
+; SCALAR-SAME: i1 [[COND:%.*]], ptr [[A:%.*]], ptr [[B:%.*]], i64 [[INDEX:%.*]]) #[[ATTR0]] {
+; SCALAR-NEXT:  [[ENTRY:.*:]]
+; SCALAR-NEXT:    [[A_B:%.*]] = select i1 [[COND]], ptr [[A]], ptr [[B]]
+; SCALAR-NEXT:    [[B_GEP:%.*]] = getelementptr i8, ptr [[A_B]], i64 [[INDEX]]
+; SCALAR-NEXT:    [[Y:%.*]] = load i8, ptr [[B_GEP]], align 1
+; SCALAR-NEXT:    ret i8 [[Y]]
+;
+entry:
+  br i1 %cond, label %if.then, label %if.else
+
+if.then:
+  %a.gep = getelementptr i8, ptr %a, i64 %index
+  %x = load i8, ptr %a.gep, align 1
+  br label %exit
+
+if.else:
+  %b.gep = getelementptr i8, ptr %b, i64 %index
+  %y = load i8, ptr %b.gep, align 1
+  br label %exit
+
+exit:
+  %r = phi i8 [ %x, %if.then ], [ %y, %if.else ]
+  ret i8 %r
+}
diff --git a/llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-gather-cost.ll b/llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-gather-cost.ll
new file mode 100644
index 0000000000000..e73ea1bf03162
--- /dev/null
+++ b/llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-gather-cost.ll
@@ -0,0 +1,189 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt < %s -mtriple=x86_64-- -mattr=+avx512f -passes='simplifycfg<sink-common-insts>' -S | FileCheck %s --check-prefixes=VF16
+; RUN: opt < %s -mtriple=x86_64-- -mattr=+avx512f,+prefer-256-bit -passes='simplifycfg<sink-common-insts>' -S | FileCheck %s --check-prefixes=VF8
+
+; Whether commoning these loads is worthwhile is decided by cost, not by
+; legality. Masked loads and gathers are both available, but a gather only pays
+; off once it replaces enough separate masked loads. The wider the vector, the
+; more loads a single gather has to absorb to break even.
+define i32 @many_load_pointers(i32 %sel, i64 %index, ptr %p0, ptr %p1, ptr %p2, ptr %p3, ptr %p4, ptr %p5, ptr %p6, ptr %p7, ptr %p8, ptr %p9, ptr %p10) {
+; VF16-LABEL: define i32 @many_load_pointers(
+; VF16-SAME: i32 [[SEL:%.*]], i64 [[INDEX:%.*]], ptr [[P0:%.*]], ptr [[P1:%.*]], ptr [[P2:%.*]], ptr [[P3:%.*]], ptr [[P4:%.*]], ptr [[P5:%.*]], ptr [[P6:%.*]], ptr [[P7:%.*]], ptr [[P8:%.*]], ptr [[P9:%.*]], ptr [[P10:%.*]]) #[[ATTR0:[0-9]+]] {
+; VF16-NEXT:  [[ENTRY:.*:]]
+; VF16-NEXT:    switch i32 [[SEL]], label %[[CASE0:.*]] [
+; VF16-NEXT:      i32 1, label %[[CASE1:.*]]
+; VF16-NEXT:      i32 2, label %[[CASE2:.*]]
+; VF16-NEXT:      i32 3, label %[[CASE3:.*]]
+; VF16-NEXT:      i32 4, label %[[CASE4:.*]]
+; VF16-NEXT:      i32 5, label %[[CASE5:.*]]
+; VF16-NEXT:      i32 6, label %[[CASE6:.*]]
+; VF16-NEXT:      i32 7, label %[[CASE7:.*]]
+; VF16-NEXT:      i32 8, label %[[CASE8:.*]]
+; VF16-NEXT:      i32 9, label %[[CASE9:.*]]
+; VF16-NEXT:      i32 10, label %[[CASE10:.*]]
+; VF16-NEXT:    ]
+; VF16:       [[CASE0]]:
+; VF16-NEXT:    [[GEP0:%.*]] = getelementptr i32, ptr [[P0]], i64 [[INDEX]]
+; VF16-NEXT:    [[V0:%.*]] = load i32, ptr [[GEP0]], align 4
+; VF16-NEXT:    br label %[[EXIT:.*]]
+; VF16:       [[CASE1]]:
+; VF16-NEXT:    [[GEP1:%.*]] = getelementptr i32, ptr [[P1]], i64 [[INDEX]]
+; VF16-NEXT:    [[V1:%.*]] = load i32, ptr [[GEP1]], align 4
+; VF16-NEXT:    br label %[[EXIT]]
+; VF16:       [[CASE2]]:
+; VF16-NEXT:    [[GEP2:%.*]] = getelementptr i32, ptr [[P2]], i64 [[INDEX]]
+; VF16-NEXT:    [[V2:%.*]] = load i32, ptr [[GEP2]], align 4
+; VF16-NEXT:    br label %[[EXIT]]
+; VF16:       [[CASE3]]:
+; VF16-NEXT:    [[GEP3:%.*]] = getelementptr i32, ptr [[P3]], i64 [[INDEX]]
+; VF16-NEXT:    [[V3:%.*]] = load i32, ptr [[GEP3]], align 4
+; VF16-NEXT:    br label %[[EXIT]]
+; VF16:       [[CASE4]]:
+; VF16-NEXT:    [[GEP4:%.*]] = getelementptr i32, ptr [[P4]], i64 [[INDEX]]
+; VF16-NEXT:    [[V4:%.*]] = load i32, ptr [[GEP4]], align 4
+; VF16-NEXT:    br label %[[EXIT]]
+; VF16:       [[CASE5]]:
+; VF16-NEXT:    [[GEP5:%.*]] = getelementptr i32, ptr [[P5]], i64 [[INDEX]]
+; VF16-NEXT:    [[V5:%.*]] = load i32, ptr [[GEP5]], align 4
+; VF16-NEXT:    br label %[[EXIT]]
+; VF16:       [[CASE6]]:
+; VF16-NEXT:    [[GEP6:%.*]] = getelementptr i32, ptr [[P6]], i64 [[INDEX]]
+; VF16-NEXT:    [[V6:%.*]] = load i32, ptr [[GEP6]], align 4
+; VF16-NEXT:    br label %[[EXIT]]
+; VF16:       [[CASE7]]:
+; VF16-NEXT:    [[GEP7:%.*]] = getelementptr i32, ptr [[P7]], i64 [[INDEX]]
+; VF16-NEXT:    [[V7:%.*]] = load i32, ptr [[GEP7]], align 4
+; VF16-NEXT:    br label %[[EXIT]]
+; VF16:       [[CASE8]]:
+; VF16-NEXT:    [[GEP8:%.*]] = getelementptr i32, ptr [[P8]], i64 [[INDEX]]
+; VF16-NEXT:    [[V8:%.*]] = load i32, ptr [[GEP8]], align 4
+; VF16-NEXT:    br label %[[EXIT]]
+; VF16:       [[CASE9]]:
+; VF16-NEXT:    [[GEP9:%.*]] = getelementptr i32, ptr [[P9]], i64 [[INDEX]]
+; VF16-NEXT:    [[V9:%.*]] = load i32, ptr [[GEP9]], align 4
+; VF16-NEXT:    br label %[[EXIT]]
+; VF16:       [[CASE10]]:
+; VF16-NEXT:    [[GEP10:%.*]] = getelementptr i32, ptr [[P10]], i64 [[INDEX]]
+; VF16-NEXT:    [[V10:%.*]] = load i32, ptr [[GEP10]], align 4
+; VF16-NEXT:    br label %[[EXIT]]
+; VF16:       [[EXIT]]:
+; VF16-NEXT:    [[RESULT:%.*]] = phi i32 [ [[V0]], %[[CASE0]] ], [ [[V1]], %[[CASE1]] ], [ [[V2]], %[[CASE2]] ], [ [[V3]], %[[CASE3]] ], [ [[V4]], %[[CASE4]] ], [ [[V5]], %[[CASE5]] ], [ [[V6]], %[[CASE6]] ], [ [[V7]], %[[CASE7]] ], [ [[V8]], %[[CASE8]] ], [ [[V9]], %[[CASE9]] ], [ [[V10]], %[[CASE10]] ]
+; VF16-NEXT:    ret i32 [[RESULT]]
+;
+; VF8-LABEL: define i32 @many_load_pointers(
+; VF8-SAME: i32 [[SEL:%.*]], i64 [[INDEX:%.*]], ptr [[P0:%.*]], ptr [[P1:%.*]], ptr [[P2:%.*]], ptr [[P3:%.*]], ptr [[P4:%.*]], ptr [[P5:%.*]], ptr [[P6:%.*]], ptr [[P7:%.*]], ptr [[P8:%.*]], ptr [[P9:%.*]], ptr [[P10:%.*]]) #[[ATTR0:[0-9]+]] {
+; VF8-NEXT:  [[ENTRY:.*]]:
+; VF8-NEXT:    switch i32 [[SEL]], label %[[EXIT:.*]] [
+; VF8-NEXT:      i32 1, label %[[CASE1:.*]]
+; VF8-NEXT:      i32 2, label %[[CASE2:.*]]
+; VF8-NEXT:      i32 3, label %[[CASE3:.*]]
+; VF8-NEXT:      i32 4, label %[[CASE4:.*]]
+; VF8-NEXT:      i32 5, label %[[CASE5:.*]]
+; VF8-NEXT:      i32 6, label %[[CASE6:.*]]
+; VF8-NEXT:      i32 7, label %[[CASE7:.*]]
+; VF8-NEXT:      i32 8, label %[[CASE8:.*]]
+; VF8-NEXT:      i32 9, label %[[CASE9:.*]]
+; VF8-NEXT:      i32 10, label %[[CASE10:.*]]
+; VF8-NEXT:    ]
+; VF8:       [[CASE1]]:
+; VF8-NEXT:    br label %[[EXIT]]
+; VF8:       [[CASE2]]:
+; VF8-NEXT:    br label %[[EXIT]]
+; VF8:       [[CASE3]]:
+; VF8-NEXT:    br label %[[EXIT]]
+; VF8:       [[CASE4]]:
+; VF8-NEXT:    br label %[[EXIT]]
+; VF8:       [[CASE5]]:
+; VF8-NEXT:    br label %[[EXIT]]
+; VF8:       [[CASE6]]:
+; VF8-NEXT:    br label %[[EXIT]]
+; VF8:       [[CASE7]]:
+; VF8-NEXT:    br label %[[EXIT]]
+; VF8:       [[CASE8]]:
+; VF8-NEXT:    br label %[[EXIT]]
+; VF8:       [[CASE9]]:
+; VF8-NEXT:    br label %[[EXIT]]
+; VF8:       [[CASE10]]:
+; VF8-NEXT:    br label %[[EXIT]]
+; VF8:       [[EXIT]]:
+; VF8-NEXT:    [[P10_SINK:%.*]] = phi ptr [ [[P10]], %[[CASE10]] ], [ [[P9]], %[[CASE9]] ], [ [[P8]], %[[CASE8]] ], [ [[P7]], %[[CASE7]] ], [ [[P6]], %[[CASE6]] ], [ [[P5]], %[[CASE5]] ], [ [[P4]], %[[CASE4]] ], [ [[P3]], %[[CASE3]] ], [ [[P2]], %[[CASE2]] ], [ [[P1]], %[[CASE1]] ], [ [[P0]], %[[ENTRY]] ]
+; VF8-NEXT:    [[GEP10:%.*]] = getelementptr i32, ptr [[P10_SINK]], i64 [[INDEX]]
+; VF8-NEXT:    [[V10:%.*]] = load i32, ptr [[GEP10]], align 4
+; VF8-NEXT:    ret i32 [[V10]]
+;
+entry:
+  switch i32 %sel, label %case0 [
+  i32 1, label %case1
+  i32 2, label %case2
+  i32 ...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/222587


More information about the llvm-commits mailing list