[llvm-branch-commits] [llvm] [Transforms] Preserve wave profiles across CFG rewrites (PR #226185)
Yaxun Liu via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Mon Sep 28 18:02:51 PDT 2026
https://github.com/yxsamliu updated https://github.com/llvm/llvm-project/pull/226185
>From e521fe94847933ae82477d88e98d948d1bcb8090 Mon Sep 17 00:00:00 2001
From: "Yaxun (Sam) Liu" <yaxun.liu at amd.com>
Date: Thu, 24 Sep 2026 08:16:39 -0400
Subject: [PATCH] [Transforms] Preserve wave profiles across CFG rewrites
HIP device PGO attaches measured wave counts to IR blocks. Later CFG
rewrites can drop counts from unchanged blocks or leave stale counts
on blocks that now execute differently. Either case makes the profile
unreliable for later optimizations.
Preserve counts through switch lowering, structurization, and loop
rotation only when a block still represents the same executions. Keep
unaffected counts and their IDs even when a loop header's count must be
invalidated. Transfer branch hints only for equivalent decisions, and
avoid assigning switch weights when default traffic cannot be traced
to one edge.
---
llvm/include/llvm/IR/ProfDataUtils.h | 2 +
llvm/lib/Transforms/Scalar/StructurizeCFG.cpp | 34 +++-
.../Transforms/Utils/LoopRotationUtils.cpp | 17 ++
llvm/lib/Transforms/Utils/LowerSwitch.cpp | 123 ++++++++++++--
llvm/test/CodeGen/AMDGPU/itofp.i128.bf.ll | 12 +-
llvm/test/CodeGen/AMDGPU/itofp.i128.ll | 36 ++--
.../Transforms/LoopRotate/wave-profile.ll | 41 +++++
.../Transforms/LowerSwitch/profile-weights.ll | 157 ++++++++++++++++++
.../Transforms/LowerSwitch/wave-profile.ll | 56 +++++++
.../StructurizeCFG/branch-profile-transfer.ll | 27 +++
.../StructurizeCFG/uniformity-profile.ll | 4 +-
.../wave-profile-loop-prefix.ll | 53 ++++++
.../Transforms/StructurizeCFG/wave-profile.ll | 50 ++++++
.../Utils/LoopRotationUtilsTest.cpp | 68 ++++++++
14 files changed, 629 insertions(+), 51 deletions(-)
create mode 100644 llvm/test/Transforms/LoopRotate/wave-profile.ll
create mode 100644 llvm/test/Transforms/LowerSwitch/profile-weights.ll
create mode 100644 llvm/test/Transforms/LowerSwitch/wave-profile.ll
create mode 100644 llvm/test/Transforms/StructurizeCFG/branch-profile-transfer.ll
create mode 100644 llvm/test/Transforms/StructurizeCFG/wave-profile-loop-prefix.ll
create mode 100644 llvm/test/Transforms/StructurizeCFG/wave-profile.ll
diff --git a/llvm/include/llvm/IR/ProfDataUtils.h b/llvm/include/llvm/IR/ProfDataUtils.h
index c87467ae3fff2..4811260c210d6 100644
--- a/llvm/include/llvm/IR/ProfDataUtils.h
+++ b/llvm/include/llvm/IR/ProfDataUtils.h
@@ -105,6 +105,8 @@ class LLVM_ABI BlockWaveCountPreserver {
public:
explicit BlockWaveCountPreserver(Function &F);
+ /// Whether construction captured a valid wave profile.
+ bool hasProfile() const { return Profile.get() != nullptr; }
void invalidate(const BasicBlock &BB);
void restore();
};
diff --git a/llvm/lib/Transforms/Scalar/StructurizeCFG.cpp b/llvm/lib/Transforms/Scalar/StructurizeCFG.cpp
index 7ac899e82ecf1..c87f7088179b9 100644
--- a/llvm/lib/Transforms/Scalar/StructurizeCFG.cpp
+++ b/llvm/lib/Transforms/Scalar/StructurizeCFG.cpp
@@ -93,8 +93,10 @@ using MaybeCondBranchWeights = std::optional<class CondBranchWeights>;
class CondBranchWeights {
uint32_t TrueWeight;
uint32_t FalseWeight;
+ bool IsExpected;
- CondBranchWeights(uint32_t T, uint32_t F) : TrueWeight(T), FalseWeight(F) {}
+ CondBranchWeights(uint32_t T, uint32_t F, bool E)
+ : TrueWeight(T), FalseWeight(F), IsExpected(E) {}
public:
static MaybeCondBranchWeights tryParse(const CondBrInst &Br) {
@@ -102,7 +104,7 @@ class CondBranchWeights {
if (!extractBranchWeights(Br, T, F))
return std::nullopt;
- return CondBranchWeights(T, F);
+ return CondBranchWeights(T, F, hasBranchWeightOrigin(Br));
}
static void setMetadata(CondBrInst &Br,
@@ -110,17 +112,18 @@ class CondBranchWeights {
if (!Weights)
return;
uint32_t Arr[] = {Weights->TrueWeight, Weights->FalseWeight};
- setBranchWeights(Br, Arr, false);
+ setBranchWeights(Br, Arr, Weights->IsExpected);
}
CondBranchWeights invert() const {
- return CondBranchWeights{FalseWeight, TrueWeight};
+ return CondBranchWeights{FalseWeight, TrueWeight, IsExpected};
}
};
struct PredInfo {
Value *Pred;
MaybeCondBranchWeights Weights;
+ MDNode *Uniformity = nullptr;
};
using BBPredicates = DenseMap<BasicBlock *, PredInfo>;
@@ -281,6 +284,7 @@ class StructurizeCFG {
const TargetTransformInfo *TTI;
Function *Func;
Region *ParentRegion;
+ BlockWaveCountPreserver *WaveProfile = nullptr;
UniformityInfo *UA = nullptr;
DominatorTree *DT;
@@ -288,6 +292,7 @@ class StructurizeCFG {
SmallVector<RegionNode *, 8> Order;
BBSet Visited;
BBSet FlowSet;
+ BBSet ChangedExecutionBlocks;
// The terminator carries a profile of the block's execution, independently
// of its branch condition. Keep it when replacing the terminator of the
@@ -585,7 +590,8 @@ PredInfo StructurizeCFG::buildCondition(CondBrInst *Term, unsigned Idx,
if (Weights)
Weights = Weights->invert();
}
- return {Cond, Weights};
+ return {Cond, Weights,
+ Term->getMetadata(LLVMContext::MD_branch_uniformity_profile)};
}
/// Analyze the predecessors of each block and build up predicates
@@ -702,7 +708,11 @@ void StructurizeCFG::insertConditions(bool Loops, SSAUpdaterBulk &PhiInserter) {
if (ParentInfo.Pred) {
Term->setCondition(ParentInfo.Pred);
- CondBranchWeights::setMetadata(*Term, ParentInfo.Weights);
+ if (!ChangedExecutionBlocks.contains(Parent)) {
+ CondBranchWeights::setMetadata(*Term, ParentInfo.Weights);
+ Term->setMetadata(LLVMContext::MD_branch_uniformity_profile,
+ ParentInfo.Uniformity);
+ }
} else {
if (!Dominator.resultIsRememberedBlock())
PhiInserter.AddAvailableValue(Variable, Dominator.result(), Default);
@@ -1112,8 +1122,11 @@ std::pair<BasicBlock *, DebugLoc> StructurizeCFG::needPrefix(bool NeedEmpty) {
if (!NeedEmpty || Entry->getFirstInsertionPt() == Entry->end()) {
// An empty prefix reused as a loop header also executes on backedges.
// Its old execution profile does not describe those additional visits.
- if (NeedEmpty)
+ if (NeedEmpty) {
BlockUniformityProfiles.erase(Entry);
+ WaveProfile->invalidate(*Entry);
+ ChangedExecutionBlocks.insert(Entry);
+ }
return {Entry, DL};
}
}
@@ -1419,6 +1432,9 @@ bool StructurizeCFG::run(Region *R, DominatorTree *DT,
this->TTI = TTI;
Func = R->getEntry()->getParent();
+ BlockWaveCountPreserver SavedWaveProfile(*Func);
+ WaveProfile = &SavedWaveProfile;
+
ParentRegion = R;
orderNodes();
@@ -1440,6 +1456,9 @@ bool StructurizeCFG::run(Region *R, DominatorTree *DT,
BB->getTerminator()->setMetadata(LLVMContext::MD_block_uniformity_profile,
MD);
+ SavedWaveProfile.restore();
+ WaveProfile = nullptr;
+
// Cleanup
BlockUniformityProfiles.clear();
Order.clear();
@@ -1452,6 +1471,7 @@ bool StructurizeCFG::run(Region *R, DominatorTree *DT,
LoopPreds.clear();
LoopConds.clear();
FlowSet.clear();
+ ChangedExecutionBlocks.clear();
return true;
}
diff --git a/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp b/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
index 5a6462d85a39a..d5703276f3dde 100644
--- a/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopRotationUtils.cpp
@@ -11,6 +11,7 @@
//===----------------------------------------------------------------------===//
#include "llvm/Transforms/Utils/LoopRotationUtils.h"
+#include "llvm/ADT/SmallPtrSet.h"
#include "llvm/ADT/Statistic.h"
#include "llvm/Analysis/AssumptionCache.h"
#include "llvm/Analysis/CodeMetrics.h"
@@ -456,6 +457,12 @@ bool LoopRotate::rotateLoop(Loop *L, bool SimplifiedLatch) {
assert(L->contains(NewHeader) && !L->contains(Exit) &&
"Unable to determine loop header and exit blocks");
+ Function &F = *OrigHeader->getParent();
+ BlockWaveCountPreserver WaveProfile(F);
+ bool HasWaveProfile = WaveProfile.hasProfile();
+ if (HasWaveProfile)
+ WaveProfile.invalidate(*OrigHeader);
+
// This code assumes that the new header has exactly one predecessor.
// Remove any single-entry PHI nodes in it.
assert(NewHeader->getSinglePredecessor() &&
@@ -823,6 +830,16 @@ bool LoopRotate::rotateLoop(Loop *L, bool SimplifiedLatch) {
if (DidMerge)
RemoveRedundantDbgInstrs(PredBB);
+ if (HasWaveProfile) {
+ // The preheader clone and merged body inherit the old header's metadata.
+ OrigPreheader->getTerminator()->setMetadata(
+ LLVMContext::MD_wave_profile_block, nullptr);
+ if (DidMerge)
+ PredBB->getTerminator()->setMetadata(LLVMContext::MD_wave_profile_block,
+ nullptr);
+ WaveProfile.restore();
+ }
+
if (MSSAU && VerifyMemorySSA)
MSSAU->getMemorySSA()->verifyMemorySSA();
diff --git a/llvm/lib/Transforms/Utils/LowerSwitch.cpp b/llvm/lib/Transforms/Utils/LowerSwitch.cpp
index 9313f313cb02d..fc3afca99973c 100644
--- a/llvm/lib/Transforms/Utils/LowerSwitch.cpp
+++ b/llvm/lib/Transforms/Utils/LowerSwitch.cpp
@@ -28,6 +28,7 @@
#include "llvm/IR/InstrTypes.h"
#include "llvm/IR/Instructions.h"
#include "llvm/IR/PassManager.h"
+#include "llvm/IR/ProfDataUtils.h"
#include "llvm/IR/Value.h"
#include "llvm/InitializePasses.h"
#include "llvm/Pass.h"
@@ -147,13 +148,101 @@ void FixPhis(BasicBlock *SuccBB, BasicBlock *OrigBB, BasicBlock *NewBB,
}
}
+// A default weight describes all values outside the explicit cases. Preserve
+// a generated branch's weights only when that mass has one possible destination
+// (including bypassing the branch); splitting it would require value profiling.
+class SwitchProfile {
+ SmallVector<APInt> Values;
+ SmallVector<uint64_t> PrefixWeights;
+ uint32_t DefaultWeight = 0;
+ APInt Lower, Upper;
+ bool IsExpected;
+
+ std::pair<unsigned, unsigned> getRange(const APInt &Lo,
+ const APInt &Hi) const {
+ auto Less = [](const APInt &A, const APInt &B) { return A.slt(B); };
+ auto Begin = llvm::lower_bound(Values, Lo, Less);
+ auto End = llvm::upper_bound(Values, Hi, Less);
+ return {Begin - Values.begin(), End - Values.begin()};
+ }
+
+ bool hasDefaultValues(const APInt &Lo, const APInt &Hi) const {
+ APInt Size =
+ Hi.sext(Hi.getBitWidth() + 1) - Lo.sext(Lo.getBitWidth() + 1) + 1;
+ auto [Begin, End] = getRange(Lo, Hi);
+ return Size.ugt(End - Begin);
+ }
+
+ uint64_t getWeight(const APInt &Lo, const APInt &Hi) const {
+ auto [Begin, End] = getRange(Lo, Hi);
+ return PrefixWeights[End] - PrefixWeights[Begin];
+ }
+
+public:
+ explicit SwitchProfile(const SwitchInst &SI)
+ : Lower(APInt::getSignedMinValue(
+ SI.getCondition()->getType()->getIntegerBitWidth())),
+ Upper(APInt::getSignedMaxValue(Lower.getBitWidth())),
+ IsExpected(hasBranchWeightOrigin(SI)) {
+ SmallVector<uint32_t> Weights;
+ if (!hasValidBranchWeightMD(SI) || !extractBranchWeights(SI, Weights))
+ return;
+ DefaultWeight = Weights[0];
+ SmallVector<std::pair<APInt, uint32_t>> Cases;
+ for (auto [Index, Case] : enumerate(SI.cases()))
+ Cases.emplace_back(Case.getCaseValue()->getValue(), Weights[Index + 1]);
+ llvm::sort(Cases, [](const auto &A, const auto &B) {
+ return A.first.slt(B.first);
+ });
+ PrefixWeights.push_back(0);
+ for (const auto &[Value, Weight] : Cases) {
+ Values.push_back(Value);
+ PrefixWeights.push_back(PrefixWeights.back() + Weight);
+ }
+ }
+
+ void setBounds(const APInt &Lo, const APInt &Hi, bool DefaultUnreachable) {
+ Lower = Lo;
+ Upper = Hi;
+ if (DefaultUnreachable)
+ DefaultWeight = 0;
+ }
+
+ void setMetadata(CondBrInst &Br, const APInt &Lo, const APInt &Hi,
+ const APInt &TrueLo, const APInt &TrueHi) const {
+ if (PrefixWeights.empty())
+ return;
+ uint64_t TrueWeight = getWeight(TrueLo, TrueHi);
+ uint64_t FalseWeight = getWeight(Lo, Hi) - TrueWeight;
+ if (DefaultWeight) {
+ bool TrueDefault = hasDefaultValues(TrueLo, TrueHi);
+ bool FalseDefault =
+ (Lo.slt(TrueLo) && hasDefaultValues(Lo, TrueLo - 1)) ||
+ (TrueHi.slt(Hi) && hasDefaultValues(TrueHi + 1, Hi));
+ bool OutsideDefault =
+ (Lower.slt(Lo) && hasDefaultValues(Lower, Lo - 1)) ||
+ (Hi.slt(Upper) && hasDefaultValues(Hi + 1, Upper));
+ if (unsigned(TrueDefault) + unsigned(FalseDefault) +
+ unsigned(OutsideDefault) >
+ 1)
+ return;
+ if (TrueDefault)
+ TrueWeight += DefaultWeight;
+ else if (FalseDefault)
+ FalseWeight += DefaultWeight;
+ }
+ setFittedBranchWeights(Br, {TrueWeight, FalseWeight}, IsExpected,
+ /*ElideAllZero=*/true);
+ }
+};
+
/// Create a new leaf block for the binary lookup tree. It checks if the
/// switch's value == the case's value. If not, then it jumps to the default
/// branch. At this point in the tree, the value can't be another valid case
/// value, so the jump to the "default" branch is warranted.
BasicBlock *NewLeafBlock(CaseRange &Leaf, Value *Val, ConstantInt *LowerBound,
ConstantInt *UpperBound, BasicBlock *OrigBlock,
- BasicBlock *Default) {
+ BasicBlock *Default, const SwitchProfile &Profile) {
Function *F = OrigBlock->getParent();
BasicBlock *NewLeaf = BasicBlock::Create(Val->getContext(), "LeafBlock");
F->insert(++OrigBlock->getIterator(), NewLeaf);
@@ -191,7 +280,9 @@ BasicBlock *NewLeafBlock(CaseRange &Leaf, Value *Val, ConstantInt *LowerBound,
// Make the conditional branch...
BasicBlock *Succ = Leaf.BB;
- CondBrInst::Create(Comp, Succ, Default, NewLeaf);
+ CondBrInst *Br = CondBrInst::Create(Comp, Succ, Default, NewLeaf);
+ Profile.setMetadata(*Br, LowerBound->getValue(), UpperBound->getValue(),
+ Leaf.Low->getValue(), Leaf.High->getValue());
// Update the PHI incoming value/block for the default.
for (auto &I : Default->phis()) {
@@ -227,7 +318,8 @@ BasicBlock *SwitchConvert(CaseItr Begin, CaseItr End, ConstantInt *LowerBound,
ConstantInt *UpperBound, Value *Val,
BasicBlock *Predecessor, BasicBlock *OrigBlock,
BasicBlock *Default,
- const std::vector<IntRange> &UnreachableRanges) {
+ const std::vector<IntRange> &UnreachableRanges,
+ const SwitchProfile &Profile) {
assert(LowerBound && UpperBound && "Bounds must be initialized");
unsigned Size = End - Begin;
@@ -241,8 +333,8 @@ BasicBlock *SwitchConvert(CaseItr Begin, CaseItr End, ConstantInt *LowerBound,
FixPhis(Begin->BB, OrigBlock, Predecessor, NumMergedCases);
return Begin->BB;
}
- return NewLeafBlock(*Begin, Val, LowerBound, UpperBound, OrigBlock,
- Default);
+ return NewLeafBlock(*Begin, Val, LowerBound, UpperBound, OrigBlock, Default,
+ Profile);
}
unsigned Mid = Size / 2;
@@ -289,15 +381,17 @@ BasicBlock *SwitchConvert(CaseItr Begin, CaseItr End, ConstantInt *LowerBound,
BasicBlock *LBranch =
SwitchConvert(LHS.begin(), LHS.end(), LowerBound, NewUpperBound, Val,
- NewNode, OrigBlock, Default, UnreachableRanges);
+ NewNode, OrigBlock, Default, UnreachableRanges, Profile);
BasicBlock *RBranch =
SwitchConvert(RHS.begin(), RHS.end(), NewLowerBound, UpperBound, Val,
- NewNode, OrigBlock, Default, UnreachableRanges);
+ NewNode, OrigBlock, Default, UnreachableRanges, Profile);
F->insert(++OrigBlock->getIterator(), NewNode);
Comp->insertInto(NewNode, NewNode->end());
- CondBrInst::Create(Comp, LBranch, RBranch, NewNode);
+ CondBrInst *Br = CondBrInst::Create(Comp, LBranch, RBranch, NewNode);
+ Profile.setMetadata(*Br, LowerBound->getValue(), UpperBound->getValue(),
+ LowerBound->getValue(), Pivot.Low->getValue() - 1);
return NewNode;
}
@@ -333,7 +427,6 @@ unsigned Clusterify(CaseVector &Cases, SwitchInst *SI) {
"Cases should be strictly ascending");
if ((nextValue == currentValue + 1) && (currentBB == nextBB)) {
I->High = J->High;
- // FIXME: Combine branch weights.
} else if (++I != J) {
*I = *J;
}
@@ -352,6 +445,7 @@ void ProcessSwitchInst(SwitchInst *SI,
BasicBlock *OrigBlock = SI->getParent();
Function *F = OrigBlock->getParent();
Value *Val = SI->getCondition(); // The value we are switching on...
+ SwitchProfile Profile(*SI);
BasicBlock *Default = SI->getDefaultDest();
// Don't handle unreachable blocks. If there are successors with phis, this
@@ -508,9 +602,11 @@ void ProcessSwitchInst(SwitchInst *SI,
Val = SI->getCondition();
}
+ Profile.setBounds(LowerBound->getValue(), UpperBound->getValue(),
+ DefaultIsUnreachableFromSwitch);
BasicBlock *SwitchBlock =
SwitchConvert(Cases.begin(), Cases.end(), LowerBound, UpperBound, Val,
- OrigBlock, OrigBlock, Default, UnreachableRanges);
+ OrigBlock, OrigBlock, Default, UnreachableRanges, Profile);
// We have added incoming values for newly-created predecessors in
// NewLeafBlock(). The only meaningful work we offload to FixPhis() is to
@@ -534,6 +630,7 @@ void ProcessSwitchInst(SwitchInst *SI,
bool LowerSwitch(Function &F, LazyValueInfo *LVI, AssumptionCache *AC) {
bool Changed = false;
+ BlockWaveCountPreserver WaveProfile(F);
SmallPtrSet<BasicBlock *, 8> DeleteList;
// We use make_early_inc_range here so that we don't traverse new blocks.
@@ -545,7 +642,11 @@ bool LowerSwitch(Function &F, LazyValueInfo *LVI, AssumptionCache *AC) {
if (SwitchInst *SI = dyn_cast<SwitchInst>(Cur.getTerminator())) {
Changed = true;
+ MDNode *Uniformity =
+ SI->getMetadata(LLVMContext::MD_block_uniformity_profile);
ProcessSwitchInst(SI, DeleteList, AC, LVI);
+ Cur.getTerminator()->setMetadata(LLVMContext::MD_block_uniformity_profile,
+ Uniformity);
}
}
@@ -554,6 +655,8 @@ bool LowerSwitch(Function &F, LazyValueInfo *LVI, AssumptionCache *AC) {
DeleteDeadBlock(BB);
}
+ if (Changed)
+ WaveProfile.restore();
return Changed;
}
diff --git a/llvm/test/CodeGen/AMDGPU/itofp.i128.bf.ll b/llvm/test/CodeGen/AMDGPU/itofp.i128.bf.ll
index 4d9a7fc63331c..c75805339c93b 100644
--- a/llvm/test/CodeGen/AMDGPU/itofp.i128.bf.ll
+++ b/llvm/test/CodeGen/AMDGPU/itofp.i128.bf.ll
@@ -52,7 +52,6 @@ define bfloat @sitofp_i128_to_bf16(i128 %x) {
; SDAG-NEXT: ; %bb.4: ; %LeafBlock
; SDAG-NEXT: v_cmp_ne_u32_e32 vcc, 26, v2
; SDAG-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; SDAG-NEXT: s_cbranch_execz .LBB0_6
; SDAG-NEXT: ; %bb.5: ; %itofp-sw-default
; SDAG-NEXT: v_sub_u32_e32 v12, 0x66, v7
; SDAG-NEXT: v_sub_u32_e32 v10, 64, v12
@@ -91,7 +90,7 @@ define bfloat @sitofp_i128_to_bf16(i128 %x) {
; SDAG-NEXT: v_or_b32_e32 v8, v15, v0
; SDAG-NEXT: v_mov_b32_e32 v0, v8
; SDAG-NEXT: v_mov_b32_e32 v1, v9
-; SDAG-NEXT: .LBB0_6: ; %Flow1
+; SDAG-NEXT: ; %bb.6: ; %Flow1
; SDAG-NEXT: s_or_b64 exec, exec, s[12:13]
; SDAG-NEXT: .LBB0_7: ; %Flow2
; SDAG-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
@@ -191,7 +190,6 @@ define bfloat @sitofp_i128_to_bf16(i128 %x) {
; GISEL-NEXT: ; %bb.4: ; %LeafBlock
; GISEL-NEXT: v_cmp_ne_u32_e32 vcc, 26, v7
; GISEL-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; GISEL-NEXT: s_cbranch_execz .LBB0_6
; GISEL-NEXT: ; %bb.5: ; %itofp-sw-default
; GISEL-NEXT: v_sub_u32_e32 v4, 0x66, v5
; GISEL-NEXT: v_sub_u32_e32 v11, 64, v4
@@ -234,7 +232,7 @@ define bfloat @sitofp_i128_to_bf16(i128 %x) {
; GISEL-NEXT: v_mov_b32_e32 v1, v4
; GISEL-NEXT: v_mov_b32_e32 v2, v5
; GISEL-NEXT: v_mov_b32_e32 v3, v6
-; GISEL-NEXT: .LBB0_6: ; %Flow1
+; GISEL-NEXT: ; %bb.6: ; %Flow1
; GISEL-NEXT: s_or_b64 exec, exec, s[12:13]
; GISEL-NEXT: .LBB0_7: ; %Flow2
; GISEL-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
@@ -330,7 +328,6 @@ define bfloat @uitofp_i128_to_bf16(i128 %x) {
; SDAG-NEXT: ; %bb.4: ; %LeafBlock
; SDAG-NEXT: v_cmp_ne_u32_e32 vcc, 26, v4
; SDAG-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; SDAG-NEXT: s_cbranch_execz .LBB1_6
; SDAG-NEXT: ; %bb.5: ; %itofp-sw-default
; SDAG-NEXT: v_sub_u32_e32 v11, 0x66, v6
; SDAG-NEXT: v_sub_u32_e32 v9, 64, v11
@@ -369,7 +366,7 @@ define bfloat @uitofp_i128_to_bf16(i128 %x) {
; SDAG-NEXT: v_or_b32_e32 v7, v14, v0
; SDAG-NEXT: v_mov_b32_e32 v0, v7
; SDAG-NEXT: v_mov_b32_e32 v1, v8
-; SDAG-NEXT: .LBB1_6: ; %Flow1
+; SDAG-NEXT: ; %bb.6: ; %Flow1
; SDAG-NEXT: s_or_b64 exec, exec, s[12:13]
; SDAG-NEXT: .LBB1_7: ; %Flow2
; SDAG-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
@@ -459,7 +456,6 @@ define bfloat @uitofp_i128_to_bf16(i128 %x) {
; GISEL-NEXT: ; %bb.4: ; %LeafBlock
; GISEL-NEXT: v_cmp_ne_u32_e32 vcc, 26, v6
; GISEL-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; GISEL-NEXT: s_cbranch_execz .LBB1_6
; GISEL-NEXT: ; %bb.5: ; %itofp-sw-default
; GISEL-NEXT: v_sub_u32_e32 v4, 0x66, v5
; GISEL-NEXT: v_sub_u32_e32 v10, 64, v4
@@ -502,7 +498,7 @@ define bfloat @uitofp_i128_to_bf16(i128 %x) {
; GISEL-NEXT: v_mov_b32_e32 v1, v4
; GISEL-NEXT: v_mov_b32_e32 v2, v5
; GISEL-NEXT: v_mov_b32_e32 v3, v6
-; GISEL-NEXT: .LBB1_6: ; %Flow1
+; GISEL-NEXT: ; %bb.6: ; %Flow1
; GISEL-NEXT: s_or_b64 exec, exec, s[12:13]
; GISEL-NEXT: .LBB1_7: ; %Flow2
; GISEL-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
diff --git a/llvm/test/CodeGen/AMDGPU/itofp.i128.ll b/llvm/test/CodeGen/AMDGPU/itofp.i128.ll
index 54f16e2a586ca..73c0574ffc19d 100644
--- a/llvm/test/CodeGen/AMDGPU/itofp.i128.ll
+++ b/llvm/test/CodeGen/AMDGPU/itofp.i128.ll
@@ -52,7 +52,6 @@ define float @sitofp_i128_to_f32(i128 %x) {
; SDAG-NEXT: ; %bb.4: ; %LeafBlock
; SDAG-NEXT: v_cmp_ne_u32_e32 vcc, 26, v2
; SDAG-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; SDAG-NEXT: s_cbranch_execz .LBB0_6
; SDAG-NEXT: ; %bb.5: ; %itofp-sw-default
; SDAG-NEXT: v_sub_u32_e32 v12, 0x66, v7
; SDAG-NEXT: v_sub_u32_e32 v10, 64, v12
@@ -91,7 +90,7 @@ define float @sitofp_i128_to_f32(i128 %x) {
; SDAG-NEXT: v_or_b32_e32 v8, v15, v0
; SDAG-NEXT: v_mov_b32_e32 v0, v8
; SDAG-NEXT: v_mov_b32_e32 v1, v9
-; SDAG-NEXT: .LBB0_6: ; %Flow1
+; SDAG-NEXT: ; %bb.6: ; %Flow1
; SDAG-NEXT: s_or_b64 exec, exec, s[12:13]
; SDAG-NEXT: .LBB0_7: ; %Flow2
; SDAG-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
@@ -184,7 +183,6 @@ define float @sitofp_i128_to_f32(i128 %x) {
; GISEL-NEXT: ; %bb.4: ; %LeafBlock
; GISEL-NEXT: v_cmp_ne_u32_e32 vcc, 26, v7
; GISEL-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; GISEL-NEXT: s_cbranch_execz .LBB0_6
; GISEL-NEXT: ; %bb.5: ; %itofp-sw-default
; GISEL-NEXT: v_sub_u32_e32 v4, 0x66, v5
; GISEL-NEXT: v_sub_u32_e32 v11, 64, v4
@@ -227,7 +225,7 @@ define float @sitofp_i128_to_f32(i128 %x) {
; GISEL-NEXT: v_mov_b32_e32 v1, v4
; GISEL-NEXT: v_mov_b32_e32 v2, v5
; GISEL-NEXT: v_mov_b32_e32 v3, v6
-; GISEL-NEXT: .LBB0_6: ; %Flow1
+; GISEL-NEXT: ; %bb.6: ; %Flow1
; GISEL-NEXT: s_or_b64 exec, exec, s[12:13]
; GISEL-NEXT: .LBB0_7: ; %Flow2
; GISEL-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
@@ -316,7 +314,6 @@ define float @uitofp_i128_to_f32(i128 %x) {
; SDAG-NEXT: ; %bb.4: ; %LeafBlock
; SDAG-NEXT: v_cmp_ne_u32_e32 vcc, 26, v4
; SDAG-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; SDAG-NEXT: s_cbranch_execz .LBB1_6
; SDAG-NEXT: ; %bb.5: ; %itofp-sw-default
; SDAG-NEXT: v_sub_u32_e32 v11, 0x66, v6
; SDAG-NEXT: v_sub_u32_e32 v9, 64, v11
@@ -355,7 +352,7 @@ define float @uitofp_i128_to_f32(i128 %x) {
; SDAG-NEXT: v_or_b32_e32 v7, v14, v0
; SDAG-NEXT: v_mov_b32_e32 v0, v7
; SDAG-NEXT: v_mov_b32_e32 v1, v8
-; SDAG-NEXT: .LBB1_6: ; %Flow1
+; SDAG-NEXT: ; %bb.6: ; %Flow1
; SDAG-NEXT: s_or_b64 exec, exec, s[12:13]
; SDAG-NEXT: .LBB1_7: ; %Flow2
; SDAG-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
@@ -438,7 +435,6 @@ define float @uitofp_i128_to_f32(i128 %x) {
; GISEL-NEXT: ; %bb.4: ; %LeafBlock
; GISEL-NEXT: v_cmp_ne_u32_e32 vcc, 26, v6
; GISEL-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; GISEL-NEXT: s_cbranch_execz .LBB1_6
; GISEL-NEXT: ; %bb.5: ; %itofp-sw-default
; GISEL-NEXT: v_sub_u32_e32 v4, 0x66, v5
; GISEL-NEXT: v_sub_u32_e32 v10, 64, v4
@@ -481,7 +477,7 @@ define float @uitofp_i128_to_f32(i128 %x) {
; GISEL-NEXT: v_mov_b32_e32 v1, v4
; GISEL-NEXT: v_mov_b32_e32 v2, v5
; GISEL-NEXT: v_mov_b32_e32 v3, v6
-; GISEL-NEXT: .LBB1_6: ; %Flow1
+; GISEL-NEXT: ; %bb.6: ; %Flow1
; GISEL-NEXT: s_or_b64 exec, exec, s[12:13]
; GISEL-NEXT: .LBB1_7: ; %Flow2
; GISEL-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
@@ -581,7 +577,6 @@ define double @sitofp_i128_to_f64(i128 %x) {
; SDAG-NEXT: ; %bb.4: ; %LeafBlock
; SDAG-NEXT: v_cmp_ne_u32_e32 vcc, 55, v2
; SDAG-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; SDAG-NEXT: s_cbranch_execz .LBB2_6
; SDAG-NEXT: ; %bb.5: ; %itofp-sw-default
; SDAG-NEXT: v_sub_u32_e32 v12, 0x49, v9
; SDAG-NEXT: v_sub_u32_e32 v10, 64, v12
@@ -624,7 +619,7 @@ define double @sitofp_i128_to_f64(i128 %x) {
; SDAG-NEXT: v_mov_b32_e32 v5, v1
; SDAG-NEXT: v_mov_b32_e32 v4, v0
; SDAG-NEXT: v_mov_b32_e32 v7, v11
-; SDAG-NEXT: .LBB2_6: ; %Flow1
+; SDAG-NEXT: ; %bb.6: ; %Flow1
; SDAG-NEXT: s_or_b64 exec, exec, s[12:13]
; SDAG-NEXT: .LBB2_7: ; %Flow2
; SDAG-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
@@ -730,7 +725,6 @@ define double @sitofp_i128_to_f64(i128 %x) {
; GISEL-NEXT: ; %bb.4: ; %LeafBlock
; GISEL-NEXT: v_cmp_ne_u32_e32 vcc, 55, v7
; GISEL-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; GISEL-NEXT: s_cbranch_execz .LBB2_6
; GISEL-NEXT: ; %bb.5: ; %itofp-sw-default
; GISEL-NEXT: v_sub_u32_e32 v14, 0x49, v9
; GISEL-NEXT: v_sub_u32_e32 v10, 64, v14
@@ -775,7 +769,7 @@ define double @sitofp_i128_to_f64(i128 %x) {
; GISEL-NEXT: v_mov_b32_e32 v3, v10
; GISEL-NEXT: v_mov_b32_e32 v4, v11
; GISEL-NEXT: v_mov_b32_e32 v5, v12
-; GISEL-NEXT: .LBB2_6: ; %Flow1
+; GISEL-NEXT: ; %bb.6: ; %Flow1
; GISEL-NEXT: s_or_b64 exec, exec, s[12:13]
; GISEL-NEXT: .LBB2_7: ; %Flow2
; GISEL-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
@@ -877,7 +871,6 @@ define double @uitofp_i128_to_f64(i128 %x) {
; SDAG-NEXT: ; %bb.4: ; %LeafBlock
; SDAG-NEXT: v_cmp_ne_u32_e32 vcc, 55, v6
; SDAG-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; SDAG-NEXT: s_cbranch_execz .LBB3_6
; SDAG-NEXT: ; %bb.5: ; %itofp-sw-default
; SDAG-NEXT: v_sub_u32_e32 v11, 0x49, v8
; SDAG-NEXT: v_sub_u32_e32 v9, 64, v11
@@ -920,7 +913,7 @@ define double @uitofp_i128_to_f64(i128 %x) {
; SDAG-NEXT: v_mov_b32_e32 v0, v4
; SDAG-NEXT: v_mov_b32_e32 v1, v5
; SDAG-NEXT: v_mov_b32_e32 v3, v10
-; SDAG-NEXT: .LBB3_6: ; %Flow1
+; SDAG-NEXT: ; %bb.6: ; %Flow1
; SDAG-NEXT: s_or_b64 exec, exec, s[12:13]
; SDAG-NEXT: .LBB3_7: ; %Flow2
; SDAG-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
@@ -1013,7 +1006,6 @@ define double @uitofp_i128_to_f64(i128 %x) {
; GISEL-NEXT: ; %bb.4: ; %LeafBlock
; GISEL-NEXT: v_cmp_ne_u32_e32 vcc, 55, v6
; GISEL-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; GISEL-NEXT: s_cbranch_execz .LBB3_6
; GISEL-NEXT: ; %bb.5: ; %itofp-sw-default
; GISEL-NEXT: v_sub_u32_e32 v13, 0x49, v8
; GISEL-NEXT: v_sub_u32_e32 v9, 64, v13
@@ -1059,7 +1051,7 @@ define double @uitofp_i128_to_f64(i128 %x) {
; GISEL-NEXT: v_mov_b32_e32 v1, v9
; GISEL-NEXT: v_mov_b32_e32 v2, v10
; GISEL-NEXT: v_mov_b32_e32 v3, v11
-; GISEL-NEXT: .LBB3_6: ; %Flow1
+; GISEL-NEXT: ; %bb.6: ; %Flow1
; GISEL-NEXT: s_or_b64 exec, exec, s[12:13]
; GISEL-NEXT: .LBB3_7: ; %Flow2
; GISEL-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
@@ -1174,7 +1166,6 @@ define half @sitofp_i128_to_f16(i128 %x) {
; SDAG-NEXT: ; %bb.4: ; %LeafBlock
; SDAG-NEXT: v_cmp_ne_u32_e32 vcc, 26, v2
; SDAG-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; SDAG-NEXT: s_cbranch_execz .LBB4_6
; SDAG-NEXT: ; %bb.5: ; %itofp-sw-default
; SDAG-NEXT: v_sub_u32_e32 v12, 0x66, v7
; SDAG-NEXT: v_sub_u32_e32 v10, 64, v12
@@ -1213,7 +1204,7 @@ define half @sitofp_i128_to_f16(i128 %x) {
; SDAG-NEXT: v_or_b32_e32 v8, v15, v0
; SDAG-NEXT: v_mov_b32_e32 v0, v8
; SDAG-NEXT: v_mov_b32_e32 v1, v9
-; SDAG-NEXT: .LBB4_6: ; %Flow1
+; SDAG-NEXT: ; %bb.6: ; %Flow1
; SDAG-NEXT: s_or_b64 exec, exec, s[12:13]
; SDAG-NEXT: .LBB4_7: ; %Flow2
; SDAG-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
@@ -1307,7 +1298,6 @@ define half @sitofp_i128_to_f16(i128 %x) {
; GISEL-NEXT: ; %bb.4: ; %LeafBlock
; GISEL-NEXT: v_cmp_ne_u32_e32 vcc, 26, v7
; GISEL-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; GISEL-NEXT: s_cbranch_execz .LBB4_6
; GISEL-NEXT: ; %bb.5: ; %itofp-sw-default
; GISEL-NEXT: v_sub_u32_e32 v4, 0x66, v5
; GISEL-NEXT: v_sub_u32_e32 v11, 64, v4
@@ -1350,7 +1340,7 @@ define half @sitofp_i128_to_f16(i128 %x) {
; GISEL-NEXT: v_mov_b32_e32 v1, v4
; GISEL-NEXT: v_mov_b32_e32 v2, v5
; GISEL-NEXT: v_mov_b32_e32 v3, v6
-; GISEL-NEXT: .LBB4_6: ; %Flow1
+; GISEL-NEXT: ; %bb.6: ; %Flow1
; GISEL-NEXT: s_or_b64 exec, exec, s[12:13]
; GISEL-NEXT: .LBB4_7: ; %Flow2
; GISEL-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
@@ -1440,7 +1430,6 @@ define half @uitofp_i128_to_f16(i128 %x) {
; SDAG-NEXT: ; %bb.4: ; %LeafBlock
; SDAG-NEXT: v_cmp_ne_u32_e32 vcc, 26, v4
; SDAG-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; SDAG-NEXT: s_cbranch_execz .LBB5_6
; SDAG-NEXT: ; %bb.5: ; %itofp-sw-default
; SDAG-NEXT: v_sub_u32_e32 v11, 0x66, v6
; SDAG-NEXT: v_sub_u32_e32 v9, 64, v11
@@ -1479,7 +1468,7 @@ define half @uitofp_i128_to_f16(i128 %x) {
; SDAG-NEXT: v_or_b32_e32 v7, v14, v0
; SDAG-NEXT: v_mov_b32_e32 v0, v7
; SDAG-NEXT: v_mov_b32_e32 v1, v8
-; SDAG-NEXT: .LBB5_6: ; %Flow1
+; SDAG-NEXT: ; %bb.6: ; %Flow1
; SDAG-NEXT: s_or_b64 exec, exec, s[12:13]
; SDAG-NEXT: .LBB5_7: ; %Flow2
; SDAG-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
@@ -1563,7 +1552,6 @@ define half @uitofp_i128_to_f16(i128 %x) {
; GISEL-NEXT: ; %bb.4: ; %LeafBlock
; GISEL-NEXT: v_cmp_ne_u32_e32 vcc, 26, v6
; GISEL-NEXT: s_and_saveexec_b64 s[12:13], vcc
-; GISEL-NEXT: s_cbranch_execz .LBB5_6
; GISEL-NEXT: ; %bb.5: ; %itofp-sw-default
; GISEL-NEXT: v_sub_u32_e32 v4, 0x66, v5
; GISEL-NEXT: v_sub_u32_e32 v10, 64, v4
@@ -1606,7 +1594,7 @@ define half @uitofp_i128_to_f16(i128 %x) {
; GISEL-NEXT: v_mov_b32_e32 v1, v4
; GISEL-NEXT: v_mov_b32_e32 v2, v5
; GISEL-NEXT: v_mov_b32_e32 v3, v6
-; GISEL-NEXT: .LBB5_6: ; %Flow1
+; GISEL-NEXT: ; %bb.6: ; %Flow1
; GISEL-NEXT: s_or_b64 exec, exec, s[12:13]
; GISEL-NEXT: .LBB5_7: ; %Flow2
; GISEL-NEXT: s_andn2_saveexec_b64 s[4:5], s[10:11]
diff --git a/llvm/test/Transforms/LoopRotate/wave-profile.ll b/llvm/test/Transforms/LoopRotate/wave-profile.ll
new file mode 100644
index 0000000000000..bbec159506ea2
--- /dev/null
+++ b/llvm/test/Transforms/LoopRotate/wave-profile.ll
@@ -0,0 +1,41 @@
+; RUN: opt -S -passes='loop-mssa(loop-rotate),verify' %s | FileCheck %s
+
+; The old header disappears. Keep the original identity space and the counts
+; for the blocks whose executions have not changed.
+
+define amdgpu_kernel void @_Z14divergent_loopPVi(ptr addrspace(1) %out) !wave.profile !0 {
+; CHECK-LABEL: define amdgpu_kernel void @_Z14divergent_loopPVi(
+; CHECK-SAME: !wave.profile [[PROFILE:![0-9]+]] {
+; CHECK: entry:
+; CHECK: br label %body, !wave.profile.block [[ENTRY:![0-9]+]]
+; CHECK: exit:
+; CHECK: ret void, !wave.profile.block [[EXIT:![0-9]+]]
+; CHECK: body:
+; CHECK: br i1 %cond, label %body, label %exit, !wave.profile.block [[BODY:![0-9]+]]
+;
+; CHECK: [[PROFILE]] = distinct !{i64 2, i64 2480672276464841217, i64 6, i64 30, i64 6, i64 24}
+; CHECK: [[ENTRY]] = !{i64 2, i64 2480672276464841217, i64 0, i64 1, i64 3}
+; CHECK: [[EXIT]] = !{i64 2, i64 2480672276464841217, i64 2, i64 1}
+; CHECK: [[BODY]] = !{i64 2, i64 2480672276464841217, i64 3, i64 1, i64 3, i64 2}
+entry:
+ br label %header, !wave.profile.block !1
+
+header:
+ %i = phi i32 [ 0, %entry ], [ %inc, %body ]
+ %cond = icmp ult i32 %i, 4
+ br i1 %cond, label %body, label %exit, !wave.profile.block !2
+
+exit:
+ ret void, !wave.profile.block !3
+
+body:
+ store volatile i32 %i, ptr addrspace(1) %out
+ %inc = add nuw nsw i32 %i, 1
+ br label %header, !wave.profile.block !4
+}
+
+!0 = !{i64 2, i64 2480672276464841217, i64 6, i64 30, i64 6, i64 24}
+!1 = !{i64 2, i64 2480672276464841217, i64 0, i64 1, i64 1}
+!2 = !{i64 2, i64 2480672276464841217, i64 1, i64 1, i64 3, i64 2}
+!3 = !{i64 2, i64 2480672276464841217, i64 2, i64 1}
+!4 = !{i64 2, i64 2480672276464841217, i64 3, i64 1, i64 1}
diff --git a/llvm/test/Transforms/LowerSwitch/profile-weights.ll b/llvm/test/Transforms/LowerSwitch/profile-weights.ll
new file mode 100644
index 0000000000000..3640e5f086496
--- /dev/null
+++ b/llvm/test/Transforms/LowerSwitch/profile-weights.ll
@@ -0,0 +1,157 @@
+; RUN: opt -S -passes='lower-switch,verify' %s | FileCheck %s
+
+; With no default traffic, each comparison receives the sum of its cases.
+; CHECK-LABEL: define void @zero_default(
+; CHECK: br i1 %Pivot{{.*}}, label {{.*}}, label {{.*}}, !prof [[ROOT:![0-9]+]]
+; CHECK: br i1 %Pivot{{.*}}, label {{.*}}, label {{.*}}, !prof [[RIGHT:![0-9]+]]
+define void @zero_default(i32 %v) {
+entry:
+ switch i32 %v, label %exit [i32 0, label %a
+ i32 1, label %b
+ i32 2, label %c], !prof !0
+ a: call void @use(i32 0)
+ br label %exit
+ b: call void @use(i32 1)
+ br label %exit
+ c: call void @use(i32 2)
+ br label %exit
+exit: ret void
+}
+
+; The default values can reach either side of the tree. Their split is unknown.
+; CHECK-LABEL: define void @unknown_default_split(
+; CHECK-NOT: !prof
+; CHECK: ret void
+define void @unknown_default_split(i32 %v) {
+entry:
+ switch i32 %v, label %exit [i32 0, label %a
+ i32 2, label %b], !prof !1
+ a: call void @use(i32 0)
+ br label %exit
+ b: call void @use(i32 1)
+ br label %exit
+exit: ret void
+}
+
+; A single comparison can retain both weights, including the expected marker.
+; CHECK-LABEL: define void @single_case(
+; CHECK: br i1 %SwitchLeaf, label %a, label %exit, !prof [[SINGLE:![0-9]+]]
+define void @single_case(i32 %v) {
+entry:
+ switch i32 %v, label %exit [i32 7, label %a], !prof !2
+ a: call void @use(i32 0)
+ br label %exit
+exit: ret void
+}
+
+; An i2 has only four values, so all default traffic must take value 1.
+; CHECK-LABEL: define void @known_default_value(
+; CHECK: br i1 %Pivot{{.*}}, label {{.*}}, label {{.*}}, !prof [[KNOWN:![0-9]+]]
+define void @known_default_value(i2 %v) {
+entry:
+ switch i2 %v, label %exit [i2 -2, label %a
+ i2 -1, label %b
+ i2 0, label %c], !prof !3
+ a: call void @use(i32 0)
+ br label %exit
+ b: call void @use(i32 1)
+ br label %exit
+ c: call void @use(i32 2)
+ br label %exit
+exit: ret void
+}
+
+; Replacing an unreachable default with a popular destination retains all of
+; that destination's case weights, including noncontiguous case values.
+; CHECK-LABEL: define void @popular_default(
+; CHECK: br i1 %SwitchLeaf, label %b, label %a, !prof [[POPULAR:![0-9]+]]
+define void @popular_default(i32 %v) {
+entry:
+ switch i32 %v, label %dead [i32 0, label %a
+ i32 1, label %a
+ i32 2, label %b
+ i32 3, label %a], !prof !4
+ a: call void @use(i32 0)
+ br label %exit
+ b: call void @use(i32 1)
+ br label %exit
+exit: ret void
+dead: unreachable
+}
+
+; Explicit cases targeting the default still have individually known weights.
+; CHECK-LABEL: define void @explicit_default_case(
+; CHECK: br i1 %Pivot{{.*}}, label {{.*}}, label {{.*}}, !prof [[POPULAR]]
+define void @explicit_default_case(i2 %v) {
+entry:
+ switch i2 %v, label %exit [i2 -2, label %a
+ i2 -1, label %exit
+ i2 0, label %b], !prof !3
+ a: call void @use(i32 0)
+ br label %exit
+ b: call void @use(i32 1)
+ br label %exit
+exit: ret void
+}
+
+; Adjacent values sharing a destination retain their combined mass.
+; CHECK-LABEL: define void @clustered_cases(
+; CHECK: br i1 %Pivot{{.*}}, label {{.*}}, label {{.*}}, !prof [[POPULAR]]
+define void @clustered_cases(i32 %v) {
+entry:
+ switch i32 %v, label %exit [i32 0, label %a
+ i32 1, label %a
+ i32 3, label %b
+ i32 4, label %c], !prof !4
+ a: call void @use(i32 0)
+ br label %exit
+ b: call void @use(i32 1)
+ br label %exit
+ c: call void @use(i32 2)
+ br label %exit
+exit: ret void
+}
+
+; Sums use 64 bits and are scaled together when they do not fit in i32.
+; CHECK-LABEL: define void @large_weights(
+; CHECK: br i1 %Pivot{{.*}}, label {{.*}}, label {{.*}}, !prof [[LARGE:![0-9]+]]
+define void @large_weights(i32 %v) {
+entry:
+ switch i32 %v, label %exit [i32 0, label %a
+ i32 1, label %b
+ i32 2, label %c], !prof !5
+ a: call void @use(i32 0)
+ br label %exit
+ b: call void @use(i32 1)
+ br label %exit
+ c: call void @use(i32 2)
+ br label %exit
+exit: ret void
+}
+
+; CHECK-LABEL: define void @all_zero(
+; CHECK-NOT: !prof
+; CHECK: ret void
+define void @all_zero(i32 %v) {
+entry:
+ switch i32 %v, label %exit [i32 7, label %a], !prof !6
+ a: call void @use(i32 0)
+ br label %exit
+exit: ret void
+}
+
+declare void @use(i32)
+!0 = !{!"branch_weights", i32 0, i32 10, i32 20, i32 30}
+!1 = !{!"branch_weights", i32 100, i32 10, i32 20}
+!2 = !{!"branch_weights", !"expected", i32 40, i32 60}
+!3 = !{!"branch_weights", i32 40, i32 10, i32 20, i32 30}
+!4 = !{!"branch_weights", i32 0, i32 10, i32 20, i32 30, i32 40}
+!5 = !{!"branch_weights", i32 0, i32 -1, i32 -1, i32 -1}
+!6 = !{!"branch_weights", i32 0, i32 0}
+
+; CHECK-DAG: [[ROOT]] = !{!"branch_weights", i32 10, i32 50}
+; CHECK-DAG: [[RIGHT]] = !{!"branch_weights", i32 20, i32 30}
+; CHECK-DAG: [[SINGLE]] = !{!"branch_weights", !"expected", i32 60, i32 40}
+; CHECK-DAG: [[KNOWN]] = !{!"branch_weights", i32 10, i32 90}
+; CHECK-DAG: [[POPULAR]] = !{!"branch_weights", i32 30, i32 70}
+; CHECK-DAG: [[LARGE]] = !{!"branch_weights", i32 2147483647, i32 -1}
diff --git a/llvm/test/Transforms/LowerSwitch/wave-profile.ll b/llvm/test/Transforms/LowerSwitch/wave-profile.ll
new file mode 100644
index 0000000000000..2b625f90e4b7c
--- /dev/null
+++ b/llvm/test/Transforms/LowerSwitch/wave-profile.ll
@@ -0,0 +1,56 @@
+; RUN: opt -S -passes='lower-switch,verify' %s | FileCheck %s --implicit-check-not=block.uniformity.profile
+; RUN: opt -S -passes='lower-switch,structurizecfg,verify' %s | FileCheck %s --check-prefix=PIPE
+
+; Keep the original invocation anchor even though original entry ID zero is
+; absent. A synthetic comparison receives an identity but no measured count.
+define void @if_else(i32 %v, ptr %out) !uniformity.profile !5 !wave.profile !0 {
+; CHECK-LABEL: define void @if_else(
+; CHECK-SAME: !wave.profile [[PROFILE:![0-9]+]]
+entry:
+; CHECK: entry:
+; CHECK: br label %LeafBlock, !block.uniformity.profile {{![0-9]+}}, !wave.profile.block [[ENTRY:![0-9]+]]
+ switch i32 %v, label %right [i32 0, label %left], !wave.profile.block !1, !block.uniformity.profile !5
+; CHECK: LeafBlock:
+; CHECK: br i1 %SwitchLeaf, label %left, label %right, !wave.profile.block [[LEAF:![0-9]+]]
+left:
+ store volatile i32 1, ptr %out
+ br label %exit, !wave.profile.block !2
+right:
+ store volatile i32 2, ptr %out
+ br label %exit, !wave.profile.block !3
+exit:
+ ret void, !wave.profile.block !4
+}
+!0 = !{i64 2, i64 2685589004101179296, i64 11, i64 10, i64 7, i64 3, i64 10}
+!1 = !{i64 2, i64 2685589004101179296, i64 1, i64 1, i64 3, i64 2}
+!2 = !{i64 2, i64 2685589004101179296, i64 2, i64 1, i64 4}
+!3 = !{i64 2, i64 2685589004101179296, i64 3, i64 1, i64 4}
+!4 = !{i64 2, i64 2685589004101179296, i64 4, i64 1}
+!5 = !{}
+; CHECK-DAG: [[PROFILE]] = distinct !{i64 2, i64 2685589004101179296, i64 11, i64 10, i64 7, i64 3, i64 10, i64 0}
+; CHECK-DAG: [[ENTRY]] = !{i64 2, i64 2685589004101179296, i64 1, i64 1, i64 5}
+; CHECK-DAG: [[LEAF]] = !{i64 2, i64 2685589004101179296, i64 5, i64 0, i64 2, i64 3}
+
+; The second CFG pass must retain measured identities and the original entry
+; anchor while assigning another unmeasured identity to its flow block.
+; PIPE-LABEL: define void @if_else(
+; PIPE-SAME: !wave.profile [[PIPE_PROFILE:![0-9]+]] {
+; PIPE: entry:
+; PIPE: br label %LeafBlock, !block.uniformity.profile {{![0-9]+}}, !wave.profile.block [[PIPE_ENTRY:![0-9]+]]
+; PIPE: LeafBlock:
+; PIPE: br i1 {{.*}}, label %right, label %Flow, !wave.profile.block [[PIPE_LEAF:![0-9]+]]
+; PIPE: Flow:
+; PIPE: br i1 {{.*}}, label %left, label %exit, !wave.profile.block [[PIPE_FLOW:![0-9]+]]
+; PIPE: left:
+; PIPE: br label %exit, !wave.profile.block [[PIPE_LEFT:![0-9]+]]
+; PIPE: right:
+; PIPE: br label %Flow, !wave.profile.block [[PIPE_RIGHT:![0-9]+]]
+; PIPE: exit:
+; PIPE: ret void, !wave.profile.block [[PIPE_EXIT:![0-9]+]]
+; PIPE: [[PIPE_PROFILE]] = distinct !{i64 2, i64 2685589004101179296, i64 11, i64 10, i64 7, i64 3, i64 10, i64 0, i64 0}
+; PIPE-DAG: [[PIPE_ENTRY]] = !{i64 2, i64 2685589004101179296, i64 1, i64 1, i64 5}
+; PIPE-DAG: [[PIPE_LEAF]] = !{i64 2, i64 2685589004101179296, i64 5, i64 0, i64 3, i64 6}
+; PIPE-DAG: [[PIPE_FLOW]] = !{i64 2, i64 2685589004101179296, i64 6, i64 0, i64 2, i64 4}
+; PIPE-DAG: [[PIPE_LEFT]] = !{i64 2, i64 2685589004101179296, i64 2, i64 1, i64 4}
+; PIPE-DAG: [[PIPE_RIGHT]] = !{i64 2, i64 2685589004101179296, i64 3, i64 1, i64 6}
+; PIPE-DAG: [[PIPE_EXIT]] = !{i64 2, i64 2685589004101179296, i64 4, i64 1}
diff --git a/llvm/test/Transforms/StructurizeCFG/branch-profile-transfer.ll b/llvm/test/Transforms/StructurizeCFG/branch-profile-transfer.ll
new file mode 100644
index 0000000000000..6bda7830e1074
--- /dev/null
+++ b/llvm/test/Transforms/StructurizeCFG/branch-profile-transfer.ll
@@ -0,0 +1,27 @@
+; RUN: opt -S -passes='structurizecfg,verify' %s | FileCheck %s --implicit-check-not=branch.uniformity.profile
+
+; Inverting the decision in the same block preserves its uniformity and swaps
+; its weights, including the llvm.expect origin. The PHI-driven Flow decision
+; combines different incoming execution populations and receives neither hint.
+; CHECK-LABEL: define void @diamond(
+; CHECK: entry:
+; CHECK: br i1 {{.*}}, label %right, label %Flow, !prof [[WEIGHTS:![0-9]+]], !branch.uniformity.profile [[UNIFORM:![0-9]+]]
+; CHECK: Flow:
+; CHECK: br i1 {{.*}}, label %left, label %exit{{$}}
+define void @diamond(i1 %c, ptr %p) !uniformity.profile !1 {
+entry:
+ br i1 %c, label %left, label %right, !prof !0, !branch.uniformity.profile !1
+left:
+ store volatile i32 1, ptr %p
+ br label %exit
+right:
+ store volatile i32 2, ptr %p
+ br label %exit
+exit:
+ ret void
+}
+
+!0 = !{!"branch_weights", !"expected", i32 90, i32 10}
+!1 = !{}
+; CHECK-DAG: [[WEIGHTS]] = !{!"branch_weights", !"expected", i32 10, i32 90}
+; CHECK-DAG: [[UNIFORM]] = !{}
diff --git a/llvm/test/Transforms/StructurizeCFG/uniformity-profile.ll b/llvm/test/Transforms/StructurizeCFG/uniformity-profile.ll
index df78540888fdf..241d62735c484 100644
--- a/llvm/test/Transforms/StructurizeCFG/uniformity-profile.ll
+++ b/llvm/test/Transforms/StructurizeCFG/uniformity-profile.ll
@@ -6,7 +6,7 @@
; CHECK-LABEL: define void @diamond(
; CHECK: entry:
-; CHECK: br i1 {{.*}}, label {{.*}}, label {{.*}}, !block.uniformity.profile ![[MD:[0-9]+]]{{$}}
+; CHECK: br i1 {{.*}}, label {{.*}}, label {{.*}}, !block.uniformity.profile ![[MD:[0-9]+]], !branch.uniformity.profile ![[MD]]{{$}}
; CHECK: then:
; CHECK-NEXT: store i32 1, ptr %out
; CHECK-NEXT: br label {{.*}}, !block.uniformity.profile ![[MD]]{{$}}
@@ -86,7 +86,7 @@ exit:
; Preserve the inner block profiles when a containing region is rewritten.
; CHECK-LABEL: define void @nested(
; CHECK: outer.then:
-; CHECK-NEXT: br i1 {{.*}}, label {{.*}}, label {{.*}}, !block.uniformity.profile ![[MD]]{{$}}
+; CHECK-NEXT: br i1 {{.*}}, label {{.*}}, label {{.*}}, !block.uniformity.profile ![[MD]], !branch.uniformity.profile ![[MD]]{{$}}
; CHECK: inner.then:
; CHECK-NEXT: store i32 5, ptr %out
; CHECK-NEXT: br label {{.*}}, !block.uniformity.profile ![[MD]]{{$}}
diff --git a/llvm/test/Transforms/StructurizeCFG/wave-profile-loop-prefix.ll b/llvm/test/Transforms/StructurizeCFG/wave-profile-loop-prefix.ll
new file mode 100644
index 0000000000000..a8eba3fd2b2d1
--- /dev/null
+++ b/llvm/test/Transforms/StructurizeCFG/wave-profile-loop-prefix.ll
@@ -0,0 +1,53 @@
+; RUN: opt -S -passes='structurizecfg,verify' %s | FileCheck %s --implicit-check-not="!prof" --implicit-check-not="!block.uniformity.profile" --implicit-check-not="!branch.uniformity.profile"
+
+; Reusing an empty prefix as a loop header changes its execution event. Keep
+; its identity but invalidate the old count, without losing the entry anchor.
+define void @if_else(i1 %enter, i1 %stop, i1 %again, ptr %out) !wave.profile !0 !uniformity.profile !9 {
+; CHECK-LABEL: define void @if_else(
+; CHECK-SAME: !wave.profile [[WAVE:![0-9]+]]
+entry:
+; CHECK: entry:
+; CHECK: br label %prefix, !wave.profile.block [[ENTRY:![0-9]+]]{{$}}
+ br label %prefix, !wave.profile.block !1
+prefix:
+; CHECK: prefix:
+; CHECK: br {{[^!]*}}!wave.profile.block [[PREFIX:![0-9]+]]{{$}}
+ br i1 %enter, label %a, label %exit.a, !prof !8, !block.uniformity.profile !9, !branch.uniformity.profile !9, !wave.profile.block !2
+a:
+; CHECK: a:
+; CHECK: br {{[^!]*}}!wave.profile.block [[A:![0-9]+]]{{$}}
+ store i32 3, ptr %out
+ br i1 %stop, label %exit.a, label %b, !wave.profile.block !3
+b:
+; CHECK: b:
+; CHECK: br {{[^!]*}}!wave.profile.block [[B:![0-9]+]]{{$}}
+ store i32 4, ptr %out
+ br i1 %again, label %a, label %exit.b, !wave.profile.block !4
+exit.a:
+ store i32 5, ptr %out
+ br label %exit, !wave.profile.block !5
+exit.b:
+ store i32 6, ptr %out
+ br label %exit, !wave.profile.block !6
+exit:
+; CHECK: exit:
+; CHECK: ret void, !wave.profile.block [[EXIT:![0-9]+]]{{$}}
+ ret void, !wave.profile.block !7
+}
+!0 = !{i64 2, i64 2685589004101179296, i64 10, i64 10, i64 30, i64 20, i64 10, i64 0, i64 10}
+!1 = !{i64 2, i64 2685589004101179296, i64 0, i64 1, i64 1}
+!2 = !{i64 2, i64 2685589004101179296, i64 1, i64 1, i64 2, i64 4}
+!3 = !{i64 2, i64 2685589004101179296, i64 2, i64 1, i64 4, i64 3}
+!4 = !{i64 2, i64 2685589004101179296, i64 3, i64 1, i64 2, i64 5}
+!5 = !{i64 2, i64 2685589004101179296, i64 4, i64 1, i64 6}
+!6 = !{i64 2, i64 2685589004101179296, i64 5, i64 1, i64 6}
+!7 = !{i64 2, i64 2685589004101179296, i64 6, i64 1}
+; CHECK: [[WAVE]] = distinct !{i64 2, i64 2685589004101179296, i64 10, i64 10, i64 30, i64 20, i64 10, i64 0, i64 10{{.*}}}
+; CHECK: [[ENTRY]] = !{i64 2, i64 2685589004101179296, i64 0, i64 1{{.*}}}
+; CHECK: [[PREFIX]] = !{i64 2, i64 2685589004101179296, i64 1, i64 0{{.*}}}
+; CHECK: [[A]] = !{i64 2, i64 2685589004101179296, i64 2, i64 1{{.*}}}
+; CHECK: [[B]] = !{i64 2, i64 2685589004101179296, i64 3, i64 1{{.*}}}
+; CHECK: [[EXIT]] = !{i64 2, i64 2685589004101179296, i64 6, i64 1}
+
+!8 = !{!"branch_weights", i32 90, i32 10}
+!9 = !{}
diff --git a/llvm/test/Transforms/StructurizeCFG/wave-profile.ll b/llvm/test/Transforms/StructurizeCFG/wave-profile.ll
new file mode 100644
index 0000000000000..b21d8e4198775
--- /dev/null
+++ b/llvm/test/Transforms/StructurizeCFG/wave-profile.ll
@@ -0,0 +1,50 @@
+; RUN: opt -S -passes='structurizecfg,verify' %s | FileCheck %s
+
+target triple = "amdgcn-amd-amdhsa"
+
+; A known CFG transform preserves a partial profile's original identities on
+; surviving blocks and refreshes their successor identities. The synthesized
+; flow block gets an identity but no measurement. The absent original entry
+; retains its normalization count (11), distinct from the current entry (10).
+
+define amdgpu_kernel void @if_else(i1 %condition, ptr addrspace(1) %out) !wave.profile !0 {
+; CHECK-LABEL: define amdgpu_kernel void @if_else(
+; CHECK-SAME: !wave.profile [[PROFILE:![0-9]+]] {
+; CHECK: entry:
+; CHECK: br i1 {{.*}}, label %{{.*}}, label %[[FLOW:.*]], {{.*}}!wave.profile.block [[ENTRY_MD:![0-9]+]]
+; CHECK: [[FLOW]]:
+; CHECK: br i1 {{.*}}, label %left, label %exit, !wave.profile.block [[FLOW_MD:![0-9]+]]
+; CHECK: left:
+; CHECK: br label %{{.*}}, !wave.profile.block [[LEFT_MD:![0-9]+]]
+; CHECK: right:
+; CHECK: br label %{{.*}}, !wave.profile.block [[RIGHT_MD:![0-9]+]]
+; CHECK: exit:
+; CHECK: ret void, !wave.profile.block [[EXIT_MD:![0-9]+]]
+;
+entry:
+ br i1 %condition, label %left, label %right, !wave.profile.block !1
+
+left:
+ store volatile i32 1, ptr addrspace(1) %out
+ br label %exit, !wave.profile.block !2
+
+right:
+ store volatile i32 2, ptr addrspace(1) %out
+ br label %exit, !wave.profile.block !3
+
+exit:
+ ret void, !wave.profile.block !4
+}
+
+!0 = !{i64 2, i64 2685589004101179296, i64 11, i64 10, i64 7, i64 3, i64 10}
+!1 = !{i64 2, i64 2685589004101179296, i64 1, i64 1, i64 2, i64 3}
+!2 = !{i64 2, i64 2685589004101179296, i64 2, i64 1, i64 4}
+!3 = !{i64 2, i64 2685589004101179296, i64 3, i64 1, i64 4}
+!4 = !{i64 2, i64 2685589004101179296, i64 4, i64 1}
+
+; CHECK: [[PROFILE]] = distinct !{i64 2, i64 2685589004101179296, i64 11, i64 10, i64 7, i64 3, i64 10, i64 0}
+; CHECK: [[ENTRY_MD]] = !{i64 2, i64 2685589004101179296, i64 1, i64 1, i64 3, i64 5}
+; CHECK: [[FLOW_MD]] = !{i64 2, i64 2685589004101179296, i64 5, i64 0, i64 2, i64 4}
+; CHECK: [[LEFT_MD]] = !{i64 2, i64 2685589004101179296, i64 2, i64 1, i64 4}
+; CHECK: [[RIGHT_MD]] = !{i64 2, i64 2685589004101179296, i64 3, i64 1, i64 5}
+; CHECK: [[EXIT_MD]] = !{i64 2, i64 2685589004101179296, i64 4, i64 1}
diff --git a/llvm/unittests/Transforms/Utils/LoopRotationUtilsTest.cpp b/llvm/unittests/Transforms/Utils/LoopRotationUtilsTest.cpp
index 70cdf0aba356c..05ebd81140f5b 100644
--- a/llvm/unittests/Transforms/Utils/LoopRotationUtilsTest.cpp
+++ b/llvm/unittests/Transforms/Utils/LoopRotationUtilsTest.cpp
@@ -17,6 +17,7 @@
#include "llvm/IR/Dominators.h"
#include "llvm/IR/LLVMContext.h"
#include "llvm/IR/Module.h"
+#include "llvm/IR/ProfDataUtils.h"
#include "llvm/Support/SourceMgr.h"
#include "gtest/gtest.h"
@@ -165,3 +166,70 @@ deopt.exit:
/// so we do change the IR.
EXPECT_TRUE(ret);
}
+
+TEST(LoopRotate, PreserveMappedWaveCounts) {
+ for (uint64_t HeaderCount : {30u, 31u}) {
+ SCOPED_TRACE(HeaderCount);
+ LLVMContext C;
+ std::unique_ptr<Module> M = parseIR(C, R"(
+define void @test(ptr %out, i32 %start) {
+entry:
+ br label %header
+header:
+ %i = phi i32 [ %start, %entry ], [ %inc, %body ]
+ %cond = icmp ult i32 %i, 4
+ br i1 %cond, label %body, label %exit
+exit:
+ ret void
+body:
+ store volatile i32 %i, ptr %out
+ %inc = add i32 %i, 1
+ br label %header
+}
+)");
+ ASSERT_TRUE(M);
+ Function &F = *M->getFunction("test");
+ if (HeaderCount == 30)
+ F.getArg(1)->replaceAllUsesWith(ConstantInt::get(Type::getInt32Ty(C), 0));
+ setBlockWaveCounts(F, {6, HeaderCount, 6, 24});
+
+ // An unused historical identity makes this a valid partial profile.
+ MDNode *Table = F.getMetadata(LLVMContext::MD_wave_profile);
+ SmallVector<Metadata *> Ops(Table->op_begin(), Table->op_end());
+ Ops.push_back(
+ ConstantAsMetadata::get(ConstantInt::get(Type::getInt64Ty(C), 0)));
+ F.setMetadata(LLVMContext::MD_wave_profile, MDNode::getDistinct(C, Ops));
+
+ SmallVector<uint64_t> Counts;
+ BitVector Valid;
+ EXPECT_FALSE(extractBlockWaveCounts(F, Counts, &Valid));
+
+ DominatorTree DT(F);
+ LoopInfo LI(DT);
+ AssumptionCache AC(F);
+ TargetTransformInfo TTI(M->getDataLayout());
+ TargetLibraryInfoImpl TLII(M->getTargetTriple());
+ TargetLibraryInfo TLI(TLII);
+ ScalarEvolution SE(F, TLI, AC, DT, LI);
+ SimplifyQuery SQ(M->getDataLayout());
+ ASSERT_TRUE(LoopRotation(*LI.begin(), &LI, &TTI, &AC, &DT, &SE, nullptr, SQ,
+ true, -1, false));
+
+ uint64_t EntryCount = 0;
+ ASSERT_TRUE(extractMappedBlockWaveCounts(F, Counts, Valid, EntryCount));
+ EXPECT_EQ(EntryCount, 6u);
+ EXPECT_GE(F.getMetadata(LLVMContext::MD_wave_profile)->getNumOperands(),
+ 7u);
+ for (auto [Index, BB] : enumerate(F)) {
+ if (BB.getName() == "entry" || BB.getName() == "exit") {
+ EXPECT_TRUE(Valid[Index]);
+ EXPECT_EQ(Counts[Index], 6u);
+ } else if (BB.getName() == "body") {
+ EXPECT_TRUE(Valid[Index]);
+ EXPECT_EQ(Counts[Index], 24u);
+ } else {
+ EXPECT_FALSE(Valid[Index]);
+ }
+ }
+ }
+}
More information about the llvm-branch-commits
mailing list