[llvm] [SimplifyCFG] Avoid sinking loads/stores that impact vectorization (PR #222587)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 10 03:23:47 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
Author: Ashutosh Nema (nema-ashutosh)
<details>
<summary>Changes</summary>
Sinking load/store memory ops through a pointer PHI forces later vectorization to use suboptimal code generation. Keep them separate when independent masked loads or stores are cheaper, matching the existing store-sinking profitability guard and applying it to loads as well.
Fixes #<!-- -->222516
---
Patch is 35.92 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/222587.diff
5 Files Affected:
- (modified) llvm/lib/Transforms/Utils/SimplifyCFG.cpp (+118-3)
- (added) llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-different-pointers.ll (+148)
- (added) llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-gather-cost.ll (+189)
- (added) llvm/test/Transforms/SimplifyCFG/X86/sink-common-store-different-pointers.ll (+152)
- (added) llvm/test/Transforms/SimplifyCFG/X86/sink-common-store-scatter-cost.ll (+190)
``````````diff
diff --git a/llvm/lib/Transforms/Utils/SimplifyCFG.cpp b/llvm/lib/Transforms/Utils/SimplifyCFG.cpp
index 6a2601487b178..9eb0d7e4bd5ee 100644
--- a/llvm/lib/Transforms/Utils/SimplifyCFG.cpp
+++ b/llvm/lib/Transforms/Utils/SimplifyCFG.cpp
@@ -2420,10 +2420,106 @@ static void sinkLastInstruction(ArrayRef<BasicBlock*> Blocks) {
}
}
+/// Estimate whether commoning \p NumMemOps loads or stores like \p I, which
+/// access distinct addresses, penalizes a later vectorizer. Commoning needs a
+/// pointer PHI, so the result can only be widened as a gather or scatter, while
+/// separate blocks allow one independent masked load or store each.
+static bool sinkingMemOpsPenalizesVectorization(const TargetTransformInfo &TTI,
+ Instruction *I,
+ unsigned NumMemOps) {
+ bool IsLoad = isa<LoadInst>(I);
+ assert((IsLoad || isa<StoreInst>(I)) && "Expected a load or store");
+ Type *ScalarTy = getLoadStoreType(I);
+ Align Alignment = getLoadStoreAlignment(I);
+ unsigned AS = getLoadStoreAddressSpace(I);
+ Intrinsic::ID MaskedID =
+ IsLoad ? Intrinsic::masked_load : Intrinsic::masked_store;
+ Intrinsic::ID GatherScatterID =
+ IsLoad ? Intrinsic::masked_gather : Intrinsic::masked_scatter;
+
+ // There is nothing to preserve unless the operations can be widened in place.
+ if (!VectorType::isValidElementType(ScalarTy) ||
+ !(IsLoad ? TTI.isLegalMaskedLoad(ScalarTy, Alignment, AS)
+ : TTI.isLegalMaskedStore(ScalarTy, Alignment, AS)))
+ return false;
+
+ // Without a gather or scatter the commoned operation cannot be widened.
+ if (!(IsLoad ? TTI.isLegalMaskedGather(ScalarTy, Alignment)
+ : TTI.isLegalMaskedScatter(ScalarTy, Alignment)))
+ return true;
+
+ TypeSize EltWidth = I->getDataLayout().getTypeSizeInBits(ScalarTy);
+ if (EltWidth.isScalable() || EltWidth.isZero())
+ return true;
+
+ auto PenalizesForVF = [&](ElementCount VF) {
+ auto *VecTy = VectorType::get(ScalarTy, VF);
+ Value *Ptr = getLoadStorePointerOperand(I);
+ auto *PtrVecTy = VectorType::get(Ptr->getType(), VF);
+ constexpr TargetTransformInfo::TargetCostKind CostKind =
+ TargetTransformInfo::TCK_RecipThroughput;
+
+ InstructionCost GatherScatterCost =
+ TTI.getAddressComputationCost(PtrVecTy, nullptr, nullptr, CostKind) +
+ TTI.getMemIntrinsicInstrCost(
+ MemIntrinsicCostAttributes(GatherScatterID, VecTy, Ptr,
+ /*VariableMask=*/true, Alignment, I),
+ CostKind);
+ InstructionCost MaskedCost =
+ TTI.getMemIntrinsicInstrCost(
+ MemIntrinsicCostAttributes(MaskedID, VecTy, Alignment, AS),
+ CostKind) *
+ NumMemOps;
+
+ if (!GatherScatterCost.isValid())
+ return true;
+ if (!MaskedCost.isValid())
+ return false;
+
+ LLVM_DEBUG(dbgs() << "SINK: " << (VF.isScalable() ? "scalable" : "fixed")
+ << " VF " << VF.getKnownMinValue() << ": "
+ << (IsLoad ? "gather" : "scatter") << " cost "
+ << GatherScatterCost << " vs " << NumMemOps << " masked "
+ << (IsLoad ? "loads " : "stores ") << MaskedCost << "\n");
+ return GatherScatterCost >= MaskedCost;
+ };
+
+ // One vector register's worth of elements, at vscale=1 for scalable vectors.
+ auto VFFrom = [&](TargetTransformInfo::RegisterKind RK,
+ bool Scalable) -> std::optional<ElementCount> {
+ TypeSize RegWidth = TTI.getRegisterBitWidth(RK);
+ if (RegWidth.isScalable() != Scalable || RegWidth.isZero())
+ return std::nullopt;
+ unsigned NumElts = RegWidth.getKnownMinValue() / EltWidth.getFixedValue();
+ if (NumElts < (Scalable ? 1u : 2u))
+ return std::nullopt;
+ return ElementCount::get(NumElts, Scalable);
+ };
+
+ std::optional<ElementCount> FixedVF =
+ VFFrom(TargetTransformInfo::RGK_FixedWidthVector, /*Scalable=*/false);
+ std::optional<ElementCount> ScalableVF =
+ VFFrom(TargetTransformInfo::RGK_ScalableVector, /*Scalable=*/true);
+
+ // Both forms are legal but no vector factor can be formed, so the target
+ // vectorizes in a way this cannot reason about. Keep the operations separate.
+ if (!FixedVF && !ScalableVF)
+ return true;
+
+ // A vectorizer may pick either kind of vector, so keep the operations
+ // separate if commoning them does not pay off for one of them.
+ bool Penalizes = false;
+ if (FixedVF)
+ Penalizes |= PenalizesForVF(*FixedVF);
+ if (ScalableVF)
+ Penalizes |= PenalizesForVF(*ScalableVF);
+ return Penalizes;
+}
+
/// Check whether BB's predecessors end with unconditional branches. If it is
/// true, sink any common code from the predecessors to BB.
-static bool sinkCommonCodeFromPredecessors(BasicBlock *BB,
- DomTreeUpdater *DTU) {
+static bool sinkCommonCodeFromPredecessors(BasicBlock *BB, DomTreeUpdater *DTU,
+ const TargetTransformInfo &TTI) {
// We support two situations:
// (1) all incoming arcs are unconditional
// (2) there are non-unconditional incoming arcs
@@ -2523,10 +2619,29 @@ static bool sinkCommonCodeFromPredecessors(BasicBlock *BB,
return false;
};
+ // Check whether the memory operations in \p Insts all access the same
+ // address, so that commoning them needs no PHI for the pointer operand.
+ auto HaveSameMemAddress = [](ArrayRef<Instruction *> Insts) {
+ Value *Ptr = getLoadStorePointerOperand(Insts.front());
+ auto *PtrI = dyn_cast<Instruction>(Ptr);
+ return all_of(drop_begin(Insts), [&](Instruction *I) {
+ Value *OtherPtr = getLoadStorePointerOperand(I);
+ if (OtherPtr == Ptr)
+ return true;
+ auto *OtherPtrI = dyn_cast<Instruction>(OtherPtr);
+ return PtrI && OtherPtrI && PtrI->isIdenticalTo(OtherPtrI);
+ });
+ };
+
// Okay, we *could* sink last ScanIdx instructions. But how many can we
// actually sink before encountering instruction that is unprofitable to
// sink?
auto ProfitableToSinkInstruction = [&](LockstepReverseIterator<true> &LRI) {
+ ArrayRef<Instruction *> Insts = *LRI;
+ if (isa<LoadInst, StoreInst>(Insts[0]) && !HaveSameMemAddress(Insts) &&
+ sinkingMemOpsPenalizesVectorization(TTI, Insts[0], Insts.size()))
+ return false;
+
unsigned NumPHIInsts = 0;
for (Use &U : (*LRI)[0]->operands()) {
auto It = PHIOperands.find(&U);
@@ -9319,7 +9434,7 @@ bool SimplifyCFGOpt::simplifyOnce(BasicBlock *BB) {
return true;
if (SinkCommon && Options.SinkCommonInsts) {
- if (sinkCommonCodeFromPredecessors(BB, DTU) ||
+ if (sinkCommonCodeFromPredecessors(BB, DTU, TTI) ||
mergeCompatibleInvokes(BB, DTU)) {
// sinkCommonCodeFromPredecessors() does not automatically CSE PHI's,
// so we may now how duplicate PHI's.
diff --git a/llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-different-pointers.ll b/llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-different-pointers.ll
new file mode 100644
index 0000000000000..5e2eb7f64c0af
--- /dev/null
+++ b/llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-different-pointers.ll
@@ -0,0 +1,148 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt < %s -mtriple=x86_64-- -mattr=+avx512f -passes='simplifycfg<sink-common-insts>' -S | FileCheck %s --check-prefixes=MASKED
+; RUN: opt < %s -mtriple=x86_64-- -mattr=+avx2 -passes='simplifycfg<sink-common-insts>' -S | FileCheck %s --check-prefixes=NOGATHER
+; RUN: opt < %s -mtriple=x86_64-- -mattr=+sse2 -passes='simplifycfg<sink-common-insts>' -S | FileCheck %s --check-prefixes=SCALAR
+
+; Loads from different address expressions can only be commoned through a
+; pointer PHI. On a target with masked loads, keep them in their original
+; blocks so that a vectorizer can widen each one on its own. AVX2 reaches the
+; same result without a gather, which cannot be widened at all.
+define i32 @different_load_pointers(i1 %cond, ptr %a, ptr %b, i64 %index) {
+; MASKED-LABEL: define i32 @different_load_pointers(
+; MASKED-SAME: i1 [[COND:%.*]], ptr [[A:%.*]], ptr [[B:%.*]], i64 [[INDEX:%.*]]) #[[ATTR0:[0-9]+]] {
+; MASKED-NEXT: [[ENTRY:.*:]]
+; MASKED-NEXT: br i1 [[COND]], label %[[IF_THEN:.*]], label %[[IF_ELSE:.*]]
+; MASKED: [[IF_THEN]]:
+; MASKED-NEXT: [[A_GEP:%.*]] = getelementptr i32, ptr [[A]], i64 [[INDEX]]
+; MASKED-NEXT: [[X:%.*]] = load i32, ptr [[A_GEP]], align 4
+; MASKED-NEXT: br label %[[EXIT:.*]]
+; MASKED: [[IF_ELSE]]:
+; MASKED-NEXT: [[B_GEP:%.*]] = getelementptr i32, ptr [[B]], i64 [[INDEX]]
+; MASKED-NEXT: [[Y:%.*]] = load i32, ptr [[B_GEP]], align 4
+; MASKED-NEXT: br label %[[EXIT]]
+; MASKED: [[EXIT]]:
+; MASKED-NEXT: [[R:%.*]] = phi i32 [ [[X]], %[[IF_THEN]] ], [ [[Y]], %[[IF_ELSE]] ]
+; MASKED-NEXT: ret i32 [[R]]
+;
+; NOGATHER-LABEL: define i32 @different_load_pointers(
+; NOGATHER-SAME: i1 [[COND:%.*]], ptr [[A:%.*]], ptr [[B:%.*]], i64 [[INDEX:%.*]]) #[[ATTR0:[0-9]+]] {
+; NOGATHER-NEXT: [[ENTRY:.*:]]
+; NOGATHER-NEXT: br i1 [[COND]], label %[[IF_THEN:.*]], label %[[IF_ELSE:.*]]
+; NOGATHER: [[IF_THEN]]:
+; NOGATHER-NEXT: [[A_GEP:%.*]] = getelementptr i32, ptr [[A]], i64 [[INDEX]]
+; NOGATHER-NEXT: [[X:%.*]] = load i32, ptr [[A_GEP]], align 4
+; NOGATHER-NEXT: br label %[[EXIT:.*]]
+; NOGATHER: [[IF_ELSE]]:
+; NOGATHER-NEXT: [[B_GEP:%.*]] = getelementptr i32, ptr [[B]], i64 [[INDEX]]
+; NOGATHER-NEXT: [[Y:%.*]] = load i32, ptr [[B_GEP]], align 4
+; NOGATHER-NEXT: br label %[[EXIT]]
+; NOGATHER: [[EXIT]]:
+; NOGATHER-NEXT: [[R:%.*]] = phi i32 [ [[X]], %[[IF_THEN]] ], [ [[Y]], %[[IF_ELSE]] ]
+; NOGATHER-NEXT: ret i32 [[R]]
+;
+; SCALAR-LABEL: define i32 @different_load_pointers(
+; SCALAR-SAME: i1 [[COND:%.*]], ptr [[A:%.*]], ptr [[B:%.*]], i64 [[INDEX:%.*]]) #[[ATTR0:[0-9]+]] {
+; SCALAR-NEXT: [[ENTRY:.*:]]
+; SCALAR-NEXT: [[A_B:%.*]] = select i1 [[COND]], ptr [[A]], ptr [[B]]
+; SCALAR-NEXT: [[B_GEP:%.*]] = getelementptr i32, ptr [[A_B]], i64 [[INDEX]]
+; SCALAR-NEXT: [[Y:%.*]] = load i32, ptr [[B_GEP]], align 4
+; SCALAR-NEXT: ret i32 [[Y]]
+;
+entry:
+ br i1 %cond, label %if.then, label %if.else
+
+if.then:
+ %a.gep = getelementptr i32, ptr %a, i64 %index
+ %x = load i32, ptr %a.gep, align 4
+ br label %exit
+
+if.else:
+ %b.gep = getelementptr i32, ptr %b, i64 %index
+ %y = load i32, ptr %b.gep, align 4
+ br label %exit
+
+exit:
+ %r = phi i32 [ %x, %if.then ], [ %y, %if.else ]
+ ret i32 %r
+}
+
+; A pointer PHI is not needed when both loads use the same address, so these
+; are commoned on any target.
+define i32 @same_load_pointer(i1 %cond, ptr %a) {
+; MASKED-LABEL: define i32 @same_load_pointer(
+; MASKED-SAME: i1 [[COND:%.*]], ptr [[A:%.*]]) #[[ATTR0]] {
+; MASKED-NEXT: [[ENTRY:.*:]]
+; MASKED-NEXT: [[X:%.*]] = load i32, ptr [[A]], align 4
+; MASKED-NEXT: ret i32 [[X]]
+;
+; NOGATHER-LABEL: define i32 @same_load_pointer(
+; NOGATHER-SAME: i1 [[COND:%.*]], ptr [[A:%.*]]) #[[ATTR0]] {
+; NOGATHER-NEXT: [[ENTRY:.*:]]
+; NOGATHER-NEXT: [[X:%.*]] = load i32, ptr [[A]], align 4
+; NOGATHER-NEXT: ret i32 [[X]]
+;
+; SCALAR-LABEL: define i32 @same_load_pointer(
+; SCALAR-SAME: i1 [[COND:%.*]], ptr [[A:%.*]]) #[[ATTR0]] {
+; SCALAR-NEXT: [[ENTRY:.*:]]
+; SCALAR-NEXT: [[X:%.*]] = load i32, ptr [[A]], align 4
+; SCALAR-NEXT: ret i32 [[X]]
+;
+entry:
+ br i1 %cond, label %if.then, label %if.else
+
+if.then:
+ %x = load i32, ptr %a, align 4
+ br label %exit
+
+if.else:
+ %y = load i32, ptr %a, align 4
+ br label %exit
+
+exit:
+ %r = phi i32 [ %x, %if.then ], [ %y, %if.else ]
+ ret i32 %r
+}
+
+; Types that the target cannot load under a mask are still commoned.
+define i8 @different_load_pointers_unsupported_type(i1 %cond, ptr %a, ptr %b, i64 %index) {
+; MASKED-LABEL: define i8 @different_load_pointers_unsupported_type(
+; MASKED-SAME: i1 [[COND:%.*]], ptr [[A:%.*]], ptr [[B:%.*]], i64 [[INDEX:%.*]]) #[[ATTR0]] {
+; MASKED-NEXT: [[ENTRY:.*:]]
+; MASKED-NEXT: [[A_B:%.*]] = select i1 [[COND]], ptr [[A]], ptr [[B]]
+; MASKED-NEXT: [[B_GEP:%.*]] = getelementptr i8, ptr [[A_B]], i64 [[INDEX]]
+; MASKED-NEXT: [[Y:%.*]] = load i8, ptr [[B_GEP]], align 1
+; MASKED-NEXT: ret i8 [[Y]]
+;
+; NOGATHER-LABEL: define i8 @different_load_pointers_unsupported_type(
+; NOGATHER-SAME: i1 [[COND:%.*]], ptr [[A:%.*]], ptr [[B:%.*]], i64 [[INDEX:%.*]]) #[[ATTR0]] {
+; NOGATHER-NEXT: [[ENTRY:.*:]]
+; NOGATHER-NEXT: [[A_B:%.*]] = select i1 [[COND]], ptr [[A]], ptr [[B]]
+; NOGATHER-NEXT: [[B_GEP:%.*]] = getelementptr i8, ptr [[A_B]], i64 [[INDEX]]
+; NOGATHER-NEXT: [[Y:%.*]] = load i8, ptr [[B_GEP]], align 1
+; NOGATHER-NEXT: ret i8 [[Y]]
+;
+; SCALAR-LABEL: define i8 @different_load_pointers_unsupported_type(
+; SCALAR-SAME: i1 [[COND:%.*]], ptr [[A:%.*]], ptr [[B:%.*]], i64 [[INDEX:%.*]]) #[[ATTR0]] {
+; SCALAR-NEXT: [[ENTRY:.*:]]
+; SCALAR-NEXT: [[A_B:%.*]] = select i1 [[COND]], ptr [[A]], ptr [[B]]
+; SCALAR-NEXT: [[B_GEP:%.*]] = getelementptr i8, ptr [[A_B]], i64 [[INDEX]]
+; SCALAR-NEXT: [[Y:%.*]] = load i8, ptr [[B_GEP]], align 1
+; SCALAR-NEXT: ret i8 [[Y]]
+;
+entry:
+ br i1 %cond, label %if.then, label %if.else
+
+if.then:
+ %a.gep = getelementptr i8, ptr %a, i64 %index
+ %x = load i8, ptr %a.gep, align 1
+ br label %exit
+
+if.else:
+ %b.gep = getelementptr i8, ptr %b, i64 %index
+ %y = load i8, ptr %b.gep, align 1
+ br label %exit
+
+exit:
+ %r = phi i8 [ %x, %if.then ], [ %y, %if.else ]
+ ret i8 %r
+}
diff --git a/llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-gather-cost.ll b/llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-gather-cost.ll
new file mode 100644
index 0000000000000..e73ea1bf03162
--- /dev/null
+++ b/llvm/test/Transforms/SimplifyCFG/X86/sink-common-load-gather-cost.ll
@@ -0,0 +1,189 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt < %s -mtriple=x86_64-- -mattr=+avx512f -passes='simplifycfg<sink-common-insts>' -S | FileCheck %s --check-prefixes=VF16
+; RUN: opt < %s -mtriple=x86_64-- -mattr=+avx512f,+prefer-256-bit -passes='simplifycfg<sink-common-insts>' -S | FileCheck %s --check-prefixes=VF8
+
+; Whether commoning these loads is worthwhile is decided by cost, not by
+; legality. Masked loads and gathers are both available, but a gather only pays
+; off once it replaces enough separate masked loads. The wider the vector, the
+; more loads a single gather has to absorb to break even.
+define i32 @many_load_pointers(i32 %sel, i64 %index, ptr %p0, ptr %p1, ptr %p2, ptr %p3, ptr %p4, ptr %p5, ptr %p6, ptr %p7, ptr %p8, ptr %p9, ptr %p10) {
+; VF16-LABEL: define i32 @many_load_pointers(
+; VF16-SAME: i32 [[SEL:%.*]], i64 [[INDEX:%.*]], ptr [[P0:%.*]], ptr [[P1:%.*]], ptr [[P2:%.*]], ptr [[P3:%.*]], ptr [[P4:%.*]], ptr [[P5:%.*]], ptr [[P6:%.*]], ptr [[P7:%.*]], ptr [[P8:%.*]], ptr [[P9:%.*]], ptr [[P10:%.*]]) #[[ATTR0:[0-9]+]] {
+; VF16-NEXT: [[ENTRY:.*:]]
+; VF16-NEXT: switch i32 [[SEL]], label %[[CASE0:.*]] [
+; VF16-NEXT: i32 1, label %[[CASE1:.*]]
+; VF16-NEXT: i32 2, label %[[CASE2:.*]]
+; VF16-NEXT: i32 3, label %[[CASE3:.*]]
+; VF16-NEXT: i32 4, label %[[CASE4:.*]]
+; VF16-NEXT: i32 5, label %[[CASE5:.*]]
+; VF16-NEXT: i32 6, label %[[CASE6:.*]]
+; VF16-NEXT: i32 7, label %[[CASE7:.*]]
+; VF16-NEXT: i32 8, label %[[CASE8:.*]]
+; VF16-NEXT: i32 9, label %[[CASE9:.*]]
+; VF16-NEXT: i32 10, label %[[CASE10:.*]]
+; VF16-NEXT: ]
+; VF16: [[CASE0]]:
+; VF16-NEXT: [[GEP0:%.*]] = getelementptr i32, ptr [[P0]], i64 [[INDEX]]
+; VF16-NEXT: [[V0:%.*]] = load i32, ptr [[GEP0]], align 4
+; VF16-NEXT: br label %[[EXIT:.*]]
+; VF16: [[CASE1]]:
+; VF16-NEXT: [[GEP1:%.*]] = getelementptr i32, ptr [[P1]], i64 [[INDEX]]
+; VF16-NEXT: [[V1:%.*]] = load i32, ptr [[GEP1]], align 4
+; VF16-NEXT: br label %[[EXIT]]
+; VF16: [[CASE2]]:
+; VF16-NEXT: [[GEP2:%.*]] = getelementptr i32, ptr [[P2]], i64 [[INDEX]]
+; VF16-NEXT: [[V2:%.*]] = load i32, ptr [[GEP2]], align 4
+; VF16-NEXT: br label %[[EXIT]]
+; VF16: [[CASE3]]:
+; VF16-NEXT: [[GEP3:%.*]] = getelementptr i32, ptr [[P3]], i64 [[INDEX]]
+; VF16-NEXT: [[V3:%.*]] = load i32, ptr [[GEP3]], align 4
+; VF16-NEXT: br label %[[EXIT]]
+; VF16: [[CASE4]]:
+; VF16-NEXT: [[GEP4:%.*]] = getelementptr i32, ptr [[P4]], i64 [[INDEX]]
+; VF16-NEXT: [[V4:%.*]] = load i32, ptr [[GEP4]], align 4
+; VF16-NEXT: br label %[[EXIT]]
+; VF16: [[CASE5]]:
+; VF16-NEXT: [[GEP5:%.*]] = getelementptr i32, ptr [[P5]], i64 [[INDEX]]
+; VF16-NEXT: [[V5:%.*]] = load i32, ptr [[GEP5]], align 4
+; VF16-NEXT: br label %[[EXIT]]
+; VF16: [[CASE6]]:
+; VF16-NEXT: [[GEP6:%.*]] = getelementptr i32, ptr [[P6]], i64 [[INDEX]]
+; VF16-NEXT: [[V6:%.*]] = load i32, ptr [[GEP6]], align 4
+; VF16-NEXT: br label %[[EXIT]]
+; VF16: [[CASE7]]:
+; VF16-NEXT: [[GEP7:%.*]] = getelementptr i32, ptr [[P7]], i64 [[INDEX]]
+; VF16-NEXT: [[V7:%.*]] = load i32, ptr [[GEP7]], align 4
+; VF16-NEXT: br label %[[EXIT]]
+; VF16: [[CASE8]]:
+; VF16-NEXT: [[GEP8:%.*]] = getelementptr i32, ptr [[P8]], i64 [[INDEX]]
+; VF16-NEXT: [[V8:%.*]] = load i32, ptr [[GEP8]], align 4
+; VF16-NEXT: br label %[[EXIT]]
+; VF16: [[CASE9]]:
+; VF16-NEXT: [[GEP9:%.*]] = getelementptr i32, ptr [[P9]], i64 [[INDEX]]
+; VF16-NEXT: [[V9:%.*]] = load i32, ptr [[GEP9]], align 4
+; VF16-NEXT: br label %[[EXIT]]
+; VF16: [[CASE10]]:
+; VF16-NEXT: [[GEP10:%.*]] = getelementptr i32, ptr [[P10]], i64 [[INDEX]]
+; VF16-NEXT: [[V10:%.*]] = load i32, ptr [[GEP10]], align 4
+; VF16-NEXT: br label %[[EXIT]]
+; VF16: [[EXIT]]:
+; VF16-NEXT: [[RESULT:%.*]] = phi i32 [ [[V0]], %[[CASE0]] ], [ [[V1]], %[[CASE1]] ], [ [[V2]], %[[CASE2]] ], [ [[V3]], %[[CASE3]] ], [ [[V4]], %[[CASE4]] ], [ [[V5]], %[[CASE5]] ], [ [[V6]], %[[CASE6]] ], [ [[V7]], %[[CASE7]] ], [ [[V8]], %[[CASE8]] ], [ [[V9]], %[[CASE9]] ], [ [[V10]], %[[CASE10]] ]
+; VF16-NEXT: ret i32 [[RESULT]]
+;
+; VF8-LABEL: define i32 @many_load_pointers(
+; VF8-SAME: i32 [[SEL:%.*]], i64 [[INDEX:%.*]], ptr [[P0:%.*]], ptr [[P1:%.*]], ptr [[P2:%.*]], ptr [[P3:%.*]], ptr [[P4:%.*]], ptr [[P5:%.*]], ptr [[P6:%.*]], ptr [[P7:%.*]], ptr [[P8:%.*]], ptr [[P9:%.*]], ptr [[P10:%.*]]) #[[ATTR0:[0-9]+]] {
+; VF8-NEXT: [[ENTRY:.*]]:
+; VF8-NEXT: switch i32 [[SEL]], label %[[EXIT:.*]] [
+; VF8-NEXT: i32 1, label %[[CASE1:.*]]
+; VF8-NEXT: i32 2, label %[[CASE2:.*]]
+; VF8-NEXT: i32 3, label %[[CASE3:.*]]
+; VF8-NEXT: i32 4, label %[[CASE4:.*]]
+; VF8-NEXT: i32 5, label %[[CASE5:.*]]
+; VF8-NEXT: i32 6, label %[[CASE6:.*]]
+; VF8-NEXT: i32 7, label %[[CASE7:.*]]
+; VF8-NEXT: i32 8, label %[[CASE8:.*]]
+; VF8-NEXT: i32 9, label %[[CASE9:.*]]
+; VF8-NEXT: i32 10, label %[[CASE10:.*]]
+; VF8-NEXT: ]
+; VF8: [[CASE1]]:
+; VF8-NEXT: br label %[[EXIT]]
+; VF8: [[CASE2]]:
+; VF8-NEXT: br label %[[EXIT]]
+; VF8: [[CASE3]]:
+; VF8-NEXT: br label %[[EXIT]]
+; VF8: [[CASE4]]:
+; VF8-NEXT: br label %[[EXIT]]
+; VF8: [[CASE5]]:
+; VF8-NEXT: br label %[[EXIT]]
+; VF8: [[CASE6]]:
+; VF8-NEXT: br label %[[EXIT]]
+; VF8: [[CASE7]]:
+; VF8-NEXT: br label %[[EXIT]]
+; VF8: [[CASE8]]:
+; VF8-NEXT: br label %[[EXIT]]
+; VF8: [[CASE9]]:
+; VF8-NEXT: br label %[[EXIT]]
+; VF8: [[CASE10]]:
+; VF8-NEXT: br label %[[EXIT]]
+; VF8: [[EXIT]]:
+; VF8-NEXT: [[P10_SINK:%.*]] = phi ptr [ [[P10]], %[[CASE10]] ], [ [[P9]], %[[CASE9]] ], [ [[P8]], %[[CASE8]] ], [ [[P7]], %[[CASE7]] ], [ [[P6]], %[[CASE6]] ], [ [[P5]], %[[CASE5]] ], [ [[P4]], %[[CASE4]] ], [ [[P3]], %[[CASE3]] ], [ [[P2]], %[[CASE2]] ], [ [[P1]], %[[CASE1]] ], [ [[P0]], %[[ENTRY]] ]
+; VF8-NEXT: [[GEP10:%.*]] = getelementptr i32, ptr [[P10_SINK]], i64 [[INDEX]]
+; VF8-NEXT: [[V10:%.*]] = load i32, ptr [[GEP10]], align 4
+; VF8-NEXT: ret i32 [[V10]]
+;
+entry:
+ switch i32 %sel, label %case0 [
+ i32 1, label %case1
+ i32 2, label %case2
+ i32 ...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/222587
More information about the llvm-commits
mailing list