[llvm] [SLSR] Skipping rewriting based on liveness (PR #218470)
Yoonseo Choi via llvm-commits
llvm-commits at lists.llvm.org
Thu Oct 1 21:25:38 PDT 2026
https://github.com/yoonseoch updated https://github.com/llvm/llvm-project/pull/218470
>From 983f3c757a8df3bbd213983b866f1f86af98ac3e Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Tue, 4 Aug 2026 00:07:56 +0000
Subject: [PATCH 01/13] [SLSR] Adding a cost model to avoid high regpressure
---
.../Scalar/StraightLineStrengthReduce.cpp | 169 +++++++++++++++++-
.../slsr-basis-distance-threshold.ll | 51 ++++++
2 files changed, 219 insertions(+), 1 deletion(-)
create mode 100644 llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 4fa462da9ec47..31fe051cca640 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -122,8 +122,15 @@ static cl::opt<bool>
EnablePoisonReuseGuard("enable-poison-reuse-guard", cl::init(true),
cl::desc("Enable poison-reuse guard"));
+<<<<<<< HEAD
STATISTIC(NumSCEVCandidateBasisDifferences,
"Number of candidate-basis SCEV differences computed by SLSR");
+=======
+static cl::opt<int> SLSRBasisDistanceThreshold(
+ "slsr-basis-distance-threshold", cl::init(96), cl::Hidden,
+ cl::desc("SLSR: skip rewrite if in-block distance from Basis's last "
+ "same-block use to the candidate Inst exceeds this"));
+>>>>>>> a5630b19d18c ([SLSR] Adding a cost model to avoid high regpressure)
namespace {
@@ -592,6 +599,15 @@ class StraightLineStrengthReduce {
if (auto *StrideInst = dyn_cast<Instruction>(C.Stride))
PropagateDependency(StrideInst);
};
+
+ bool hasOperandsUsedInNonRewritableUsersInAnotherBlock(
+ llvm::Instruction *Inst) const;
+ bool hasRewritableCandidates(const Instruction *Inst) const;
+ bool basisTooFarInSameBlock(
+ const Candidate &C,
+ DenseMap<const BasicBlock *, DenseMap<const Instruction *, int>>
+ &IndexCache,
+ const Instruction *Inst) const;
};
inline raw_ostream &operator<<(raw_ostream &OS,
@@ -1411,6 +1427,118 @@ bool StraightLineStrengthReduceLegacyPass::runOnFunction(Function &F) {
return StraightLineStrengthReduce(DL, DT, SE, TTI).runOnFunction(F);
}
+// Go through all operands of instruction, and check if any operand is used in
+// another block different from the instruction's block and the another use is
+// not rewritable. return true if such operand is found, otherwise return false.
+bool StraightLineStrengthReduce::
+ hasOperandsUsedInNonRewritableUsersInAnotherBlock(
+ llvm::Instruction *Inst) const {
+ llvm::BasicBlock *InstBB = Inst->getParent();
+ if (!InstBB)
+ return false;
+
+ for (Value *OpVal : Inst->operand_values()) {
+ auto *OpInst = dyn_cast<Instruction>(OpVal);
+ if (!OpInst)
+ continue;
+
+ for (const User *U : OpInst->users())
+ if (auto *UI = dyn_cast<Instruction>(U))
+ if (UI->getParent() != InstBB && !hasRewritableCandidates(UI)) {
+ LLVM_DEBUG(dbgs()
+ << "Inst's operand is used in another block "
+ << "("
+ << (InstBB->hasName() ? InstBB->getName() : "unnamed")
+ << " -> "
+ << (UI->getParent() && UI->getParent()->hasName()
+ ? UI->getParent()->getName()
+ : "unnamed")
+ << ") " << *UI << "\n");
+ return true;
+ }
+ }
+ return false;
+}
+
+bool StraightLineStrengthReduce::hasRewritableCandidates(
+ const Instruction *Inst) const {
+ if (!RewriteCandidates.count(Inst))
+ return false;
+
+ for (const Candidate *C : RewriteCandidates.at(Inst))
+ if (C->Basis)
+ return true;
+
+ return false;
+}
+
+// Assign a monotonically increasing index to each (non-debug/pseudo)
+// instruction in BB so in-block distances can be queried in O(1) once built.
+static DenseMap<const Instruction *, int>
+buildBlockIndexMap(const BasicBlock &BB) {
+ DenseMap<const Instruction *, int> IndexMap;
+ int Index = 0;
+ for (const Instruction &I : BB) {
+ // Skip debug/pseudo instructions so the distance math is identical
+ // between debug and release builds.
+ if (I.isDebugOrPseudoInst())
+ continue;
+ IndexMap[&I] = Index++;
+ }
+ return IndexMap;
+}
+
+// Return true
+// 1. if C.Basis used only in C.Ins's block before C.Ins
+// AND
+// 2. if any use of C.Basis before C.Ins and C.Ins exceeds
+// SLSRBasisDistanceThreshold.
+bool StraightLineStrengthReduce::basisTooFarInSameBlock(
+ const Candidate &C,
+ DenseMap<const BasicBlock *, DenseMap<const Instruction *, int>>
+ &IndexCache,
+ const Instruction *I) const {
+ Instruction *Inst = C.Ins;
+ assert(Inst == I);
+ Instruction *BasisInst = C.Basis ? C.Basis->Ins : nullptr;
+ if (!BasisInst)
+ return false;
+
+ const BasicBlock *BB = Inst->getParent();
+ auto [It, Inserted] = IndexCache.try_emplace(BB);
+ if (Inserted)
+ It->second = buildBlockIndexMap(*BB);
+ const DenseMap<const Instruction *, int> &IndexMap = It->second;
+
+ auto InstIt = IndexMap.find(Inst);
+ if (InstIt == IndexMap.end())
+ return false;
+ int InstIdx = InstIt->second;
+
+ int LastUseIdx = 0;
+
+ bool FoundSameBlockUse = false;
+ for (const User *U : BasisInst->users()) {
+ const auto *UI = dyn_cast<Instruction>(U);
+ if (!UI)
+ continue;
+ // If one of the uses is not in the same block, return false.
+ if (UI->getParent() != BB)
+ return false;
+ auto UseIt = IndexMap.find(UI);
+ // If any same block use is later than Inst, return false.
+ if (UseIt == IndexMap.end() || UseIt->second >= InstIdx)
+ return false;
+ FoundSameBlockUse = true;
+ LastUseIdx = std::max(LastUseIdx, UseIt->second);
+ }
+
+ if (!FoundSameBlockUse)
+ return false;
+
+ return (InstIdx - LastUseIdx) > SLSRBasisDistanceThreshold;
+}
+
bool StraightLineStrengthReduce::runOnFunction(Function &F) {
LLVM_DEBUG(dbgs() << "SLSR on Function: " << F.getName() << "\n");
// Traverse the dominator tree in the depth-first order. This order makes sure
@@ -1427,11 +1555,50 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
}
sortCandidateInstructions();
+ // From SortedCandidateInsts, remove some candidates that are likely to
+ // increase register pressure. The candidate's Inst is the source of
+ // replacement. A candidate in the following criteria should be removed:
+ // 1. The candidate's Inst "has operands used in non-rewritable users in
+ // another block"
+ // -- checked by hasOperandsUsedInNonRewritableUsersInAnotherBlock(Inst)
+ // -- This means the candidate's Inst's original operands are live in
+ // another block, so even if rewrite the Inst, the operands may be still
+ // live out to another block.
+ // -- Thus, rewriting the Inst based on Basis might add another long live
+ // range from the Basis by increasing the live range of the Basis.
+ // -- TODO: If needed, a refinement to check that "another block" is
+ // properly dominated by the candidate's Inst's block can be added.
+ // 2. When the candidate's Basis's should beused only in the the same block
+ // and its last use is before the candidate's Inst, the difference between the
+ // last use of Basis and the Inst is larger than a threshold.
+ // -- This is also for avoiding increasing the live range of the Basis by
+ // rewriting the Inst.
+ //
+ // A candidate satisfies both conditions 1 and 2 should be removed.
+
+ // Collect candidates likely to increase register pressure.
+ // Evaluate on the original IR, before any rewriteCandidate mutates it
+ // Done before rewriting: rewriting inserts instructions and does
+ // replaceAllUsesWith, which would invalidate both the in-block index map and
+ // operands' user sets.
+ DenseSet<Instruction *> ToSkipRewrite;
+ {
+ DenseMap<const BasicBlock *, DenseMap<const Instruction *, int>> IndexCache;
+ for (Instruction *I : SortedCandidateInsts)
+ if (Candidate *C = pickRewriteCandidate(I))
+ if (hasOperandsUsedInNonRewritableUsersInAnotherBlock(I) &&
+ basisTooFarInSameBlock(*C, IndexCache, I))
+ ToSkipRewrite.insert(I);
+ }
+
// Rewrite candidates in the topological order that rewrites a Candidate
// always before rewriting its Basis
- for (Instruction *I : reverse(SortedCandidateInsts))
+ for (Instruction *I : reverse(SortedCandidateInsts)) {
+ if (ToSkipRewrite.contains(I))
+ continue;
if (Candidate *C = pickRewriteCandidate(I))
rewriteCandidate(*C);
+ }
for (auto *DeadIns : DeadInstructions)
// A dead instruction may be another dead instruction's op,
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
new file mode 100644
index 0000000000000..96d05db79bd2e
--- /dev/null
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
@@ -0,0 +1,51 @@
+; RUN: opt < %s -passes=slsr -S -slsr-basis-distance-threshold=96 | FileCheck %s --check-prefixes=CHECK,REWRITE
+; RUN: opt < %s -passes=slsr -S -slsr-basis-distance-threshold=2 | FileCheck %s --check-prefixes=CHECK,SKIP
+
+; The register-pressure cost model skips a rewrite only when BOTH:
+; 1. an operand of the candidate is used by a non-rewritable user in
+; another block, and
+; 2. the basis' last same-block use is farther than
+; -slsr-basis-distance-threshold from the candidate.
+
+target datalayout = "e-i64:64-v16:16-v32:32-n16:32:64"
+
+declare void @foo(i32)
+declare void @use(i32)
+
+define void @basis_too_far(i32 %b, i32 %s) {
+; CHECK-LABEL: @basis_too_far(
+; CHECK: %t1 = add i32 %b, %s
+; CHECK: %s2 = shl i32 %s, 1
+; REWRITE: %t2 = add i32 %t1, %s
+; SKIP: %t2 = add i32 %b, %s2
+entry:
+ %t1 = add i32 %b, %s
+ call void @foo(i32 %t1)
+ call void @foo(i32 %b)
+ call void @foo(i32 %b)
+ call void @foo(i32 %b)
+ %s2 = shl i32 %s, 1
+ %t2 = add i32 %b, %s2
+ call void @foo(i32 %t2)
+ br label %next
+
+next:
+ call void @use(i32 %s2)
+ ret void
+}
+
+define void @same_block_operand(i32 %b, i32 %s) {
+; CHECK-LABEL: @same_block_operand(
+; CHECK: %t2 = add i32 %t1, %s
+entry:
+ %t1 = add i32 %b, %s
+ call void @foo(i32 %t1)
+ call void @foo(i32 %b)
+ call void @foo(i32 %b)
+ call void @foo(i32 %b)
+ %s2 = shl i32 %s, 1
+ %t2 = add i32 %b, %s2
+ call void @foo(i32 %t2)
+ call void @use(i32 %s2)
+ ret void
+}
>From aceb0d71d8a235019c7bd23ab7bab79a39aa81bb Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Wed, 12 Aug 2026 06:38:47 +0000
Subject: [PATCH 02/13] Clean-ups based on reviewers' comments; Autogen a
lit-test
---
.../Scalar/StraightLineStrengthReduce.cpp | 21 ++++---
.../slsr-basis-distance-threshold.ll | 57 ++++++++++++++++---
2 files changed, 60 insertions(+), 18 deletions(-)
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 31fe051cca640..abf14f10ad4ab 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -122,15 +122,13 @@ static cl::opt<bool>
EnablePoisonReuseGuard("enable-poison-reuse-guard", cl::init(true),
cl::desc("Enable poison-reuse guard"));
-<<<<<<< HEAD
-STATISTIC(NumSCEVCandidateBasisDifferences,
- "Number of candidate-basis SCEV differences computed by SLSR");
-=======
static cl::opt<int> SLSRBasisDistanceThreshold(
"slsr-basis-distance-threshold", cl::init(96), cl::Hidden,
cl::desc("SLSR: skip rewrite if in-block distance from Basis's last "
"same-block use to the candidate Inst exceeds this"));
->>>>>>> a5630b19d18c ([SLSR] Adding a cost model to avoid high regpressure)
+
+STATISTIC(NumSCEVCandidateBasisDifferences,
+ "Number of candidate-basis SCEV differences computed by SLSR");
namespace {
@@ -1428,14 +1426,12 @@ bool StraightLineStrengthReduceLegacyPass::runOnFunction(Function &F) {
}
// Go through all operands of instruction, and check if any operand is used in
-// another block different from the instruction's block and the another use is
+// another block different from the instruction's block and the other use is
// not rewritable. return true if such operand is found, otherwise return false.
bool StraightLineStrengthReduce::
hasOperandsUsedInNonRewritableUsersInAnotherBlock(
llvm::Instruction *Inst) const {
llvm::BasicBlock *InstBB = Inst->getParent();
- if (!InstBB)
- return false;
for (Value *OpVal : Inst->operand_values()) {
auto *OpInst = dyn_cast<Instruction>(OpVal);
@@ -1443,7 +1439,9 @@ bool StraightLineStrengthReduce::
continue;
for (const User *U : OpInst->users())
- if (auto *UI = dyn_cast<Instruction>(U))
+ if (auto *UI = dyn_cast<Instruction>(U)) {
+ if (UI->isDebugOrPseudoInst())
+ continue;
if (UI->getParent() != InstBB && !hasRewritableCandidates(UI)) {
LLVM_DEBUG(dbgs()
<< "Inst's operand is used in another block "
@@ -1456,6 +1454,7 @@ bool StraightLineStrengthReduce::
<< ") " << *UI << "\n");
return true;
}
+ }
}
return false;
}
@@ -1543,7 +1542,7 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
LLVM_DEBUG(dbgs() << "SLSR on Function: " << F.getName() << "\n");
// Traverse the dominator tree in the depth-first order. This order makes sure
// all bases of a candidate are in Candidates when we process it.
- for (const auto Node : depth_first(DT))
+ for (auto *const Node : depth_first(DT))
for (auto &I : *(Node->getBlock()))
allocateCandidatesAndFindBasis(&I);
@@ -1568,7 +1567,7 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
// range from the Basis by increasing the live range of the Basis.
// -- TODO: If needed, a refinement to check that "another block" is
// properly dominated by the candidate's Inst's block can be added.
- // 2. When the candidate's Basis's should beused only in the the same block
+ // 2. When the candidate's Basis's is only used in the the same block
// and its last use is before the candidate's Inst, the difference between the
// last use of Basis and the Inst is larger than a threshold.
// -- This is also for avoiding increasing the live range of the Basis by
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
index 96d05db79bd2e..42aa95b8dd006 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
@@ -1,3 +1,4 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
; RUN: opt < %s -passes=slsr -S -slsr-basis-distance-threshold=96 | FileCheck %s --check-prefixes=CHECK,REWRITE
; RUN: opt < %s -passes=slsr -S -slsr-basis-distance-threshold=2 | FileCheck %s --check-prefixes=CHECK,SKIP
@@ -6,6 +7,9 @@
; another block, and
; 2. the basis' last same-block use is farther than
; -slsr-basis-distance-threshold from the candidate.
+;
+; Rewriting of %t2 = add i32 %t1, %s based on basis %t1 shouldn't happen when slsr-basis-distance-threshold is 2.
+; Distance from %t1's last use to %t2 is 5 > 2 and %s2 continues to be used in "next".
target datalayout = "e-i64:64-v16:16-v32:32-n16:32:64"
@@ -13,11 +17,38 @@ declare void @foo(i32)
declare void @use(i32)
define void @basis_too_far(i32 %b, i32 %s) {
-; CHECK-LABEL: @basis_too_far(
-; CHECK: %t1 = add i32 %b, %s
-; CHECK: %s2 = shl i32 %s, 1
-; REWRITE: %t2 = add i32 %t1, %s
-; SKIP: %t2 = add i32 %b, %s2
+; REWRITE-LABEL: define void @basis_too_far(
+; REWRITE-SAME: i32 [[B:%.*]], i32 [[S:%.*]]) {
+; REWRITE-NEXT: [[ENTRY:.*:]]
+; REWRITE-NEXT: [[T1:%.*]] = add i32 [[B]], [[S]]
+; REWRITE-NEXT: call void @foo(i32 [[T1]])
+; REWRITE-NEXT: call void @foo(i32 [[B]])
+; REWRITE-NEXT: call void @foo(i32 [[B]])
+; REWRITE-NEXT: call void @foo(i32 [[B]])
+; REWRITE-NEXT: [[S2:%.*]] = shl i32 [[S]], 1
+; REWRITE-NEXT: [[T2:%.*]] = add i32 [[T1]], [[S]]
+; REWRITE-NEXT: call void @foo(i32 [[T2]])
+; REWRITE-NEXT: br label %[[NEXT:.*]]
+; REWRITE: [[NEXT]]:
+; REWRITE-NEXT: call void @use(i32 [[S2]])
+; REWRITE-NEXT: ret void
+;
+; SKIP-LABEL: define void @basis_too_far(
+; SKIP-SAME: i32 [[B:%.*]], i32 [[S:%.*]]) {
+; SKIP-NEXT: [[ENTRY:.*:]]
+; SKIP-NEXT: [[T1:%.*]] = add i32 [[B]], [[S]]
+; SKIP-NEXT: call void @foo(i32 [[T1]])
+; SKIP-NEXT: call void @foo(i32 [[B]])
+; SKIP-NEXT: call void @foo(i32 [[B]])
+; SKIP-NEXT: call void @foo(i32 [[B]])
+; SKIP-NEXT: [[S2:%.*]] = shl i32 [[S]], 1
+; SKIP-NEXT: [[T2:%.*]] = add i32 [[B]], [[S2]]
+; SKIP-NEXT: call void @foo(i32 [[T2]])
+; SKIP-NEXT: br label %[[NEXT:.*]]
+; SKIP: [[NEXT]]:
+; SKIP-NEXT: call void @use(i32 [[S2]])
+; SKIP-NEXT: ret void
+;
entry:
%t1 = add i32 %b, %s
call void @foo(i32 %t1)
@@ -35,8 +66,20 @@ next:
}
define void @same_block_operand(i32 %b, i32 %s) {
-; CHECK-LABEL: @same_block_operand(
-; CHECK: %t2 = add i32 %t1, %s
+; CHECK-LABEL: define void @same_block_operand(
+; CHECK-SAME: i32 [[B:%.*]], i32 [[S:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[T1:%.*]] = add i32 [[B]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T1]])
+; CHECK-NEXT: call void @foo(i32 [[B]])
+; CHECK-NEXT: call void @foo(i32 [[B]])
+; CHECK-NEXT: call void @foo(i32 [[B]])
+; CHECK-NEXT: [[S2:%.*]] = shl i32 [[S]], 1
+; CHECK-NEXT: [[T2:%.*]] = add i32 [[T1]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T2]])
+; CHECK-NEXT: call void @use(i32 [[S2]])
+; CHECK-NEXT: ret void
+;
entry:
%t1 = add i32 %b, %s
call void @foo(i32 %t1)
>From cd17f6cf22f0b932e903d77bd735317392170b44 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Sat, 22 Aug 2026 22:08:44 +0000
Subject: [PATCH 03/13] Liveness-based RPFilter
---
.../Scalar/StraightLineStrengthReduce.cpp | 411 ++++++++++++++++++
1 file changed, 411 insertions(+)
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index abf14f10ad4ab..da75915ec3e94 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -71,6 +71,7 @@
#include "llvm/Transforms/Scalar/StraightLineStrengthReduce.h"
#include "llvm/ADT/APInt.h"
#include "llvm/ADT/DepthFirstIterator.h"
+#include "llvm/ADT/PostOrderIterator.h"
#include "llvm/ADT/SetVector.h"
#include "llvm/ADT/SmallPtrSet.h"
#include "llvm/ADT/SmallVector.h"
@@ -110,6 +111,7 @@ using namespace llvm;
using namespace PatternMatch;
#define DEBUG_TYPE "slsr"
+#define DEBUG_SLSR_RP(X) DEBUG_WITH_TYPE(DEBUG_TYPE "-rp", X)
static const unsigned UnknownAddressSpace =
std::numeric_limits<unsigned>::max();
@@ -1538,6 +1540,395 @@ bool StraightLineStrengthReduce::basisTooFarInSameBlock(
return (InstIdx - LastUseIdx) > SLSRBasisDistanceThreshold;
}
+namespace {
+
+// TODO: Currently, I am considering (Basis, Cand) pair that are both in the
+// same BB.
+// The restriction may not be needed.
+class RPFilter {
+public:
+ using Candidate = StraightLineStrengthReduce::Candidate;
+
+ RPFilter(const Function *F,
+ DenseMap<Instruction *, Candidate *> &PickedCandidateMap)
+ //, const TargetTransformInfo *TTI)
+ : F(F), PickedCandidateMap(PickedCandidateMap) {} //, TTI(TTI) {}
+ void run() {
+ buildBBToNumCandsAndBasises(PickedCandidateMap);
+ // TODO: 16 is an arbitrary threshold.
+ if (MaxNumBasisesInBB > 16)
+ buildBBToLiveness(*F); // Do liveness analysis.
+
+ for (auto &BB : *F) {
+#if 0
+ unsigned RP = maxPressureInBlock(BB, BBToLiveness[&BB].LiveIn,
+ BBToLiveness[&BB].LiveOut);
+ unsigned RPBackward = maxPressureInBlockBackward(BB, BBToLiveness[&BB].LiveIn,
+ BBToLiveness[&BB].LiveOut);
+ DEBUG_WITH_TYPE("slsr-rp", {
+ dbgs() << "MaxRP:" << BB.getName() << ":" << RP << "," << RPBackward << "\n";
+ });
+#else
+ unsigned RPBackward = maxPressureInBlockBackward(
+ BB, BBToLiveness[&BB].LiveIn, BBToLiveness[&BB].LiveOut);
+ DEBUG_SLSR_RP(dbgs() << "MaxRPBackward:" << BB.getName() << ":"
+ << RPBackward << "\n");
+#endif
+ }
+ }
+
+private:
+ const Function *F;
+ DenseMap<Instruction *, Candidate *> &PickedCandidateMap;
+ // const TargetTransformInfo *TTI;
+
+ DenseMap<const BasicBlock *, std::pair<unsigned, unsigned>>
+ BBToNumCandsAndBasises;
+ unsigned MaxNumBasisesInBB = 0;
+
+ using ValueSet = SmallPtrSet<const Value *, 32>;
+ struct BlockLiveness {
+ ValueSet LiveIn;
+ ValueSet LiveOut;
+ };
+
+ DenseMap<const BasicBlock *, BlockLiveness> BBToLiveness;
+
+ std::pair<unsigned, unsigned> countCandsAndBasisesInBB(
+ const BasicBlock *BB,
+ const DenseMap<Instruction *, Candidate *> &PickedCandidateMap) {
+ unsigned NumCands = 0;
+ SmallPtrSet<const Candidate *, 8> UniqueBasises;
+ for (const Instruction &Inst : *BB) {
+ auto It = PickedCandidateMap.find(&Inst);
+ if (It != PickedCandidateMap.end()) {
+ NumCands++;
+ assert(It->second->Basis);
+ UniqueBasises.insert(It->second->Basis);
+ }
+ }
+ MaxNumBasisesInBB = std::max(MaxNumBasisesInBB, UniqueBasises.size());
+ DEBUG_WITH_TYPE("slsr-rp", {
+ dbgs() << "BB: " << BB->getName() << " NumCands: " << NumCands
+ << " UniqueBasises: " << UniqueBasises.size() << "\n";
+ });
+ return {NumCands, UniqueBasises.size()};
+ }
+
+ void buildBBToNumCandsAndBasises(
+ const DenseMap<Instruction *, Candidate *> &PickedCandidateMap) {
+ for (auto &InstCand : PickedCandidateMap) {
+ const BasicBlock *BB = InstCand.first->getParent();
+ auto [It, Inserted] = BBToNumCandsAndBasises.try_emplace(BB);
+ if (Inserted)
+ It->second = countCandsAndBasisesInBB(It->first, PickedCandidateMap);
+ }
+ }
+
+ static bool isRegisterLike(const Value *V) {
+ return isa<Instruction>(V) || isa<Argument>(V);
+ }
+
+ unsigned regWeight(const Value *V) const {
+ const DataLayout &DL = F->getDataLayout();
+ Type *Ty = V->getType();
+ if (Ty->isVoidTy() || Ty->isTokenTy())
+ return 0;
+#if 0
+ // Aggregates legalize to a flat sequence of scalars; approximate rather
+ // than calling getRegUsageForType, which is llvm_unreachable on them.
+ if (!VectorType::isValidElementType(Ty->getScalarType()))
+ return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
+ return TTI->getRegUsageForType(Ty);
+#else
+ // TTI's getRegUsageForType can be used for types that are
+ // validElementType(Ty->getScalarType()). However, even the valid vector
+ // type should be multiplited by get{Min}NumElements() to handle
+ // {ScalarVectorType}, FixedVectorType. Overall, getTypeSizeInBits(Ty)
+ // handing all those cases should be sufficient for heuristic.
+ return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
+#endif
+ }
+
+ void buildBBToLiveness(const Function &F) {
+ DenseMap<const BasicBlock *, ValueSet> UpExposed, Defs;
+ DenseMap<std::pair<const BasicBlock *, const BasicBlock *>, ValueSet>
+ PhiEdges;
+
+ // Fill in per-block information
+ for (const BasicBlock &BB : F) {
+ ValueSet &UE = UpExposed[&BB];
+ ValueSet &D = Defs[&BB];
+
+ for (const Instruction &I : BB) {
+ if (const auto *PN = dyn_cast<PHINode>(&I)) {
+ for (unsigned i = 0, e = PN->getNumIncomingValues(); i < e; ++i)
+ if (isRegisterLike(PN->getIncomingValue(i)))
+ PhiEdges[{PN->getIncomingBlock(i), &BB}].insert(
+ PN->getIncomingValue(i));
+ } else {
+ for (const Value *Op : I.operand_values())
+ if (isRegisterLike(Op) && !D.contains(Op))
+ UE.insert(Op);
+ }
+
+ if (!I.getType()->isVoidTy())
+ D.insert(&I);
+ }
+ DEBUG_WITH_TYPE("slsr-rp", {
+ dbgs() << "BB-fill: " << BB.getName()
+ << " UpExposed: " << UpExposed[&BB].size();
+ dbgs() << " Defs: " << Defs[&BB].size() << "\n";
+ });
+ }
+
+ // Block Live-in/out fixed-point loop
+ bool Changed = true;
+ while (Changed) {
+ Changed = false;
+ for (const BasicBlock *BB : post_order(&F.getEntryBlock())) {
+ ValueSet Out;
+ for (const BasicBlock *S : successors(BB)) {
+ // Out = Out + BBToLiveness[S].LiveIn
+ for (const Value *V : BBToLiveness[S].LiveIn)
+ Out.insert(V);
+ auto It = PhiEdges.find({BB, S});
+ if (It != PhiEdges.end())
+ // Out = Out + It->second
+ for (const Value *V : It->second)
+ Out.insert(V);
+ }
+ ValueSet In = UpExposed[BB];
+ // live-throughs are live-ins
+ for (const Value *V : Out)
+ if (!Defs[BB].contains(V))
+ In.insert(V);
+
+ if (In.size() != BBToLiveness[BB].LiveIn.size() ||
+ Out.size() != BBToLiveness[BB].LiveOut.size())
+ Changed = true;
+
+ DEBUG_WITH_TYPE("slsr-rp", {
+ dbgs() << "BB-update: " << BB->getName() << " In: " << In.size()
+ << " Out: " << Out.size() << "\n";
+ });
+
+ BBToLiveness[BB].LiveIn = std::move(In);
+ BBToLiveness[BB].LiveOut = std::move(Out);
+ }
+ }
+
+ DEBUG_WITH_TYPE("slsr-rp", {
+ dbgs() << "-- Liveness of BBs -- \n";
+ for (const BasicBlock &BB : F) {
+ dbgs() << BB.getName() << ": ";
+ dbgs() << BBToLiveness[&BB].LiveIn.size() << ", ";
+ dbgs() << BBToLiveness[&BB].LiveOut.size() << "\n";
+ }
+ });
+ }
+
+ bool isDebugBlock(const BasicBlock &BB) const {
+ return BB.getName() == "for.cond.cleanup";
+ }
+
+ unsigned maxPressureInBlock(const BasicBlock &BB, const ValueSet &LiveIn,
+ const ValueSet &LiveOut) const {
+
+#if 1
+ bool IsDebugBlock = isDebugBlock(BB);
+ unsigned StoreCount = 0;
+#endif
+
+ SmallVector<const Instruction *, 128> Order;
+ DenseMap<const Instruction *, unsigned> Idx;
+ for (const Instruction &I : BB) {
+ if (I.isDebugOrPseudoInst())
+ continue;
+ Idx[&I] = Order.size();
+ Order.push_back(&I);
+ }
+
+ DenseMap<const Value *, unsigned> LastUse;
+ for (const Instruction &I : BB) {
+ if (I.isDebugOrPseudoInst() || isa<PHINode>(&I))
+ continue;
+ for (const Value *Op : I.operand_values())
+ if (isRegisterLike(Op))
+ LastUse[Op] = Idx[&I];
+ }
+ const unsigned End = Order.size();
+ for (const Value *V : LiveOut)
+ LastUse[V] = End; // survives the block; never retires here
+
+ ValueSet Open;
+ for (const Value *V : LiveIn)
+ Open.insert(V);
+
+ unsigned MaxW = 0;
+ for (unsigned i = 0; i != End; ++i) {
+ if (isa<StoreInst>(Order[i])) {
+ StoreCount++;
+ }
+ if (!Order[i]->getType()->isVoidTy())
+ Open.insert(Order[i]);
+
+ unsigned W = 0;
+ for (const Value *V : Open) {
+ auto RW = regWeight(V);
+ W += RW;
+ if (IsDebugBlock && StoreCount == 32) {
+ DEBUG_SLSR_RP({
+ dbgs() << "regweight: " << "i: " << i << " value: " << *V
+ << " RW: " << RW << ", ";
+ });
+ // dbgs() << "W: " << W << "\n";
+ }
+
+ // W += regWeight(V);
+ }
+ MaxW = std::max(MaxW, W);
+
+ SmallVector<const Value *, 8> Dead;
+ for (const Value *V : Open) {
+ auto It = LastUse.find(V);
+ if (It == LastUse.end() || It->second <= i)
+ Dead.push_back(V);
+ }
+ for (const Value *V : Dead)
+ Open.erase(V);
+
+ if (IsDebugBlock && StoreCount == 32) {
+ DEBUG_SLSR_RP(dbgs() << "W: " << W << " MaxW: " << MaxW << "\n");
+ }
+ }
+ return MaxW;
+ }
+
+ // Return true if I is a candidate and its basis is in the same bb, false
+ // otherwise.
+ bool insertBasisIfCand(const Instruction *I, ValueSet &LiveSetWithSLSR,
+ DenseSet<const Value *> &SeenLastUse) const {
+ auto It = PickedCandidateMap.find(I);
+ if (It == PickedCandidateMap.end())
+ return false;
+
+ // I is a Candidate
+ const Instruction *Basis = It->second->Basis->Ins;
+ if (Basis->getParent() == I->getParent()) {
+ LiveSetWithSLSR.insert(Basis);
+ // I and its Basis are in the same bb.
+
+ SeenLastUse.insert(Basis);
+ return true;
+ }
+
+ return false;
+ }
+
+ // TODO: Remove
+ void updateLiveSetWithSLSR(ValueSet &LiveSetWithSLSR,
+ const DenseSet<const Value *> &SeenLastUse,
+ const Instruction &I, const Value *Op) const {
+ // LiveSet += {Cand.Basis} <-- Done already
+ // LiveSet -= {Op} if this is the last use of Op (i.e.
+ // SeenLastUse.contains(Op))
+ if (SeenLastUse.contains(Op))
+ LiveSetWithSLSR.erase(Op);
+ }
+
+ unsigned maxPressureInBlockBackward(const BasicBlock &BB,
+ const ValueSet &LiveIn,
+ const ValueSet &LiveOut) const {
+
+ // TODO: ValueSet is a SmallPtrSet, which is supposedly smaller than 33.
+ // Could be a better-fitting data structure.
+ ValueSet LiveSet = LiveOut;
+ unsigned MaxW = 0;
+ for (const Value *V : LiveSet)
+ MaxW += regWeight(V);
+
+ // Initial LiveSetWithSLSR is the same as LiveOut.
+ // SLSR changes
+ // Basis = ..
+ // ..
+ // Cand = f(op1, op2, ..)
+ //
+ // to
+ // Basis = ..
+ // Cand = f(Basis, Delta, // possibly some of the original operands ..)
+ //
+ // We consider here only the case both Cand and Basis are in the same BB.
+ // Therefore, if an original operand of Cand is a liveout, it means it was
+ // used outside the BB, and will stay so. Likewise, if a Basis is a liveout,
+ // it means it was used outside the BB, and will stay so. SLSR does not
+ // delete existing defs or create new defs.
+ ValueSet LiveSetWithSLSR = LiveOut;
+ unsigned MaxWWithSLSR = MaxW;
+
+ // This is required for calculating LiveSetWithSLSR without updating IR.
+ // IR is still in the original form.
+ // Intialize it with LiveOut. LiveOuts are never removed from
+ // LiveSetWithSLSR.
+ DenseSet<const Value *> SeenLastUse(LiveOut.begin(), LiveOut.end());
+
+ // Scan BB backward
+ for (const Instruction &I : reverse(BB)) {
+ if (I.isDebugOrPseudoInst() || isa<PHINode>(&I))
+ continue;
+
+ bool IsCand = insertBasisIfCand(&I, LiveSetWithSLSR, SeenLastUse);
+ for (const Value *Op : I.operand_values())
+ if (isRegisterLike(Op)) {
+ LiveSet.insert(Op);
+ if (IsCand && !SeenLastUse.contains(Op)) {
+ // Op is the last use. It will be replaced by Basis, so it isremoved
+ // from LiveSetWithSLSR. As a heuristic, we don't dicern which Op of
+ // I is replaced by Basis. When IsCand is true, there is only
+ // reg-like one Op in practice.
+ // TODO: Can further restrict by Candidate's type and DeltaKind.
+ LiveSetWithSLSR.erase(Op);
+ DEBUG_SLSR_RP(dbgs() << "Removed Op from LiveSetWithSLSR: " << *Op
+ << " in inst " << I << "\n");
+ } else {
+ LiveSetWithSLSR.insert(Op);
+ }
+ // Op's insertion happens only if Op was not there before.
+ // This logic works as BB is scanned backward.
+ SeenLastUse.insert(Op);
+ }
+
+ // Compute RP reaching for this instruction.
+ unsigned W = 0;
+ for (const Value *V : LiveSet) {
+ // TODO: Cache regWeight(V)
+ auto RW = regWeight(V);
+ W += RW;
+ }
+ MaxW = std::max(MaxW, W);
+ W = 0;
+ for (const Value *V : LiveSetWithSLSR) {
+ // TODO: Cache regWeight(V)
+ auto RW = regWeight(V);
+ W += RW;
+ }
+ MaxWWithSLSR = std::max(MaxWWithSLSR, W);
+
+ // Remove the def (Instruction) from LiveSet for the next upward
+ // instruction.
+ if (!I.getType()->isVoidTy()) {
+ LiveSet.erase(&I);
+ LiveSetWithSLSR.erase(&I);
+ }
+ }
+
+ DEBUG_SLSR_RP(dbgs() << "MaxWWithSLSR: " << MaxWWithSLSR << "\n");
+
+ return MaxW;
+ }
+};
+} // end of anonymous namespace
+
bool StraightLineStrengthReduce::runOnFunction(Function &F) {
LLVM_DEBUG(dbgs() << "SLSR on Function: " << F.getName() << "\n");
// Traverse the dominator tree in the depth-first order. This order makes sure
@@ -1554,6 +1945,26 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
}
sortCandidateInstructions();
+ ////////////////////////////////////////////////////
+ DenseMap<Instruction *, Candidate *> PickedCandidateMap;
+ for (Instruction *I : SortedCandidateInsts)
+ if (Candidate *C = pickRewriteCandidate(I))
+ PickedCandidateMap[I] = C;
+
+ // RPFilter RPFilter(&F, PickedCandidateMap, TTI);
+ RPFilter RPFilter(&F, PickedCandidateMap);
+ RPFilter.run();
+
+#if 0
+ DenseMap<const BasicBlock*, unsigned> BBToNumCands;
+ for (auto &InstCand : PickedCandidateMap) {
+ const BasicBlock *BB= InstCand.first->getParent();
+ auto [It, Inserted] = BBToNumCands.try_emplace(BB);
+ if (Inserted)
+ It->second = countCandsInBB(It->first, PickedCandidateMap);
+ }
+#endif
+
// From SortedCandidateInsts, remove some candidates that are likely to
// increase register pressure. The candidate's Inst is the source of
// replacement. A candidate in the following criteria should be removed:
>From e6a11a1f824899f6bbd1d8be4535d13362a57793 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Tue, 25 Aug 2026 00:10:03 +0000
Subject: [PATCH 04/13] Adding an option to use TTI's getRegUsageFor as an
alternative
---
.../Scalar/StraightLineStrengthReduce.cpp | 90 +++++++++++--------
1 file changed, 52 insertions(+), 38 deletions(-)
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index da75915ec3e94..8fb8dced13072 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -112,6 +112,7 @@ using namespace PatternMatch;
#define DEBUG_TYPE "slsr"
#define DEBUG_SLSR_RP(X) DEBUG_WITH_TYPE(DEBUG_TYPE "-rp", X)
+#define DEBUG_SLSR_RP_DETAIL(X) DEBUG_WITH_TYPE(DEBUG_TYPE "-rp-detail", X)
static const unsigned UnknownAddressSpace =
std::numeric_limits<unsigned>::max();
@@ -129,6 +130,9 @@ static cl::opt<int> SLSRBasisDistanceThreshold(
cl::desc("SLSR: skip rewrite if in-block distance from Basis's last "
"same-block use to the candidate Inst exceeds this"));
+static cl::opt<bool> UseTTIForRP("slsr-tti-rp", cl::init(false),
+ cl::desc("Use TTI to compute RP for SLSR-RP"));
+
STATISTIC(NumSCEVCandidateBasisDifferences,
"Number of candidate-basis SCEV differences computed by SLSR");
@@ -1550,15 +1554,17 @@ class RPFilter {
using Candidate = StraightLineStrengthReduce::Candidate;
RPFilter(const Function *F,
- DenseMap<Instruction *, Candidate *> &PickedCandidateMap)
- //, const TargetTransformInfo *TTI)
- : F(F), PickedCandidateMap(PickedCandidateMap) {} //, TTI(TTI) {}
+ DenseMap<Instruction *, Candidate *> &PickedCandidateMap,
+ const TargetTransformInfo *TTI)
+ : F(F), PickedCandidateMap(PickedCandidateMap), TTI(TTI) {}
+
void run() {
buildBBToNumCandsAndBasises(PickedCandidateMap);
// TODO: 16 is an arbitrary threshold.
if (MaxNumBasisesInBB > 16)
buildBBToLiveness(*F); // Do liveness analysis.
+ DEBUG_SLSR_RP(dbgs() << "-- MaxRP of BBs -- \n");
for (auto &BB : *F) {
#if 0
unsigned RP = maxPressureInBlock(BB, BBToLiveness[&BB].LiveIn,
@@ -1569,10 +1575,10 @@ class RPFilter {
dbgs() << "MaxRP:" << BB.getName() << ":" << RP << "," << RPBackward << "\n";
});
#else
- unsigned RPBackward = maxPressureInBlockBackward(
+ auto [MaxRP, MaxRPWithSLSR] = maxPressureInBlockBackward(
BB, BBToLiveness[&BB].LiveIn, BBToLiveness[&BB].LiveOut);
- DEBUG_SLSR_RP(dbgs() << "MaxRPBackward:" << BB.getName() << ":"
- << RPBackward << "\n");
+ DEBUG_SLSR_RP(dbgs() << "MaxRP:" << BB.getName() << ": (" << MaxRP << ", "
+ << MaxRPWithSLSR << ")" << "\n");
#endif
}
}
@@ -1580,7 +1586,7 @@ class RPFilter {
private:
const Function *F;
DenseMap<Instruction *, Candidate *> &PickedCandidateMap;
- // const TargetTransformInfo *TTI;
+ const TargetTransformInfo *TTI;
DenseMap<const BasicBlock *, std::pair<unsigned, unsigned>>
BBToNumCandsAndBasises;
@@ -1609,7 +1615,7 @@ class RPFilter {
}
MaxNumBasisesInBB = std::max(MaxNumBasisesInBB, UniqueBasises.size());
DEBUG_WITH_TYPE("slsr-rp", {
- dbgs() << "BB: " << BB->getName() << " NumCands: " << NumCands
+ dbgs() << "BB: " << BB->getName() << " - NumCands: " << NumCands
<< " UniqueBasises: " << UniqueBasises.size() << "\n";
});
return {NumCands, UniqueBasises.size()};
@@ -1634,20 +1640,30 @@ class RPFilter {
Type *Ty = V->getType();
if (Ty->isVoidTy() || Ty->isTokenTy())
return 0;
+ if (UseTTIForRP) {
+ // Aggregates legalize to a flat sequence of scalars; approximate rather
+ // than calling getRegUsageForType, which is llvm_unreachable on them.
+ if (!VectorType::isValidElementType(Ty->getScalarType()))
+ return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
+
+ // TTI's getRegUsageForType can be used for types that are
+ // validElementType(Ty->getScalarType()). However, even the valid vector
+ // type should be multiplited by get{Min}NumElements() to handle
+ // {ScalarVectorType}, FixedVectorType. Overall, getTypeSizeInBits(Ty)
+ // handing all those cases should be sufficient for heuristic.
+ unsigned RegUsage = TTI->getRegUsageForType(Ty);
#if 0
- // Aggregates legalize to a flat sequence of scalars; approximate rather
- // than calling getRegUsageForType, which is llvm_unreachable on them.
- if (!VectorType::isValidElementType(Ty->getScalarType()))
- return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
- return TTI->getRegUsageForType(Ty);
-#else
- // TTI's getRegUsageForType can be used for types that are
- // validElementType(Ty->getScalarType()). However, even the valid vector
- // type should be multiplited by get{Min}NumElements() to handle
- // {ScalarVectorType}, FixedVectorType. Overall, getTypeSizeInBits(Ty)
- // handing all those cases should be sufficient for heuristic.
- return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
+ if (FixedVectorType *FVT = dyn_cast<FixedVectorType>(Ty))
+ RegUsage *= FVT->getNumElements();
+ else if (ScalableVectorType *SVT = dyn_cast<ScalableVectorType>(Ty))
+ RegUsage *= SVT->getMinNumElements();
#endif
+ return RegUsage;
+
+ } else {
+ // Default logic to compute RP.
+ return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
+ }
}
void buildBBToLiveness(const Function &F) {
@@ -1675,7 +1691,7 @@ class RPFilter {
if (!I.getType()->isVoidTy())
D.insert(&I);
}
- DEBUG_WITH_TYPE("slsr-rp", {
+ DEBUG_SLSR_RP_DETAIL({
dbgs() << "BB-fill: " << BB.getName()
<< " UpExposed: " << UpExposed[&BB].size();
dbgs() << " Defs: " << Defs[&BB].size() << "\n";
@@ -1708,7 +1724,7 @@ class RPFilter {
Out.size() != BBToLiveness[BB].LiveOut.size())
Changed = true;
- DEBUG_WITH_TYPE("slsr-rp", {
+ DEBUG_SLSR_RP_DETAIL({
dbgs() << "BB-update: " << BB->getName() << " In: " << In.size()
<< " Out: " << Out.size() << "\n";
});
@@ -1719,7 +1735,7 @@ class RPFilter {
}
DEBUG_WITH_TYPE("slsr-rp", {
- dbgs() << "-- Liveness of BBs -- \n";
+ dbgs() << "-- Live Ins/Outs of BBs -- \n";
for (const BasicBlock &BB : F) {
dbgs() << BB.getName() << ": ";
dbgs() << BBToLiveness[&BB].LiveIn.size() << ", ";
@@ -1837,9 +1853,9 @@ class RPFilter {
LiveSetWithSLSR.erase(Op);
}
- unsigned maxPressureInBlockBackward(const BasicBlock &BB,
- const ValueSet &LiveIn,
- const ValueSet &LiveOut) const {
+ std::pair<unsigned, unsigned>
+ maxPressureInBlockBackward(const BasicBlock &BB, const ValueSet &LiveIn,
+ const ValueSet &LiveOut) const {
// TODO: ValueSet is a SmallPtrSet, which is supposedly smaller than 33.
// Could be a better-fitting data structure.
@@ -1882,19 +1898,20 @@ class RPFilter {
if (isRegisterLike(Op)) {
LiveSet.insert(Op);
if (IsCand && !SeenLastUse.contains(Op)) {
- // Op is the last use. It will be replaced by Basis, so it isremoved
- // from LiveSetWithSLSR. As a heuristic, we don't dicern which Op of
- // I is replaced by Basis. When IsCand is true, there is only
- // reg-like one Op in practice.
+ // Op is the last use. It will be replaced by Basis, so it is
+ // removed from LiveSetWithSLSR. As a heuristic, we don't dicern
+ // which Op of I is replaced by Basis. When IsCand is true, there is
+ // only one reg-like Op in practice.
// TODO: Can further restrict by Candidate's type and DeltaKind.
LiveSetWithSLSR.erase(Op);
- DEBUG_SLSR_RP(dbgs() << "Removed Op from LiveSetWithSLSR: " << *Op
- << " in inst " << I << "\n");
+ DEBUG_SLSR_RP_DETAIL(dbgs() << "Removed Op from LiveSetWithSLSR: "
+ << *Op << " in inst " << I << "\n");
} else {
LiveSetWithSLSR.insert(Op);
}
// Op's insertion happens only if Op was not there before.
- // This logic works as BB is scanned backward.
+ // Therefore, SeenLastUse keeps the last use of Op as
+ // BB is scanned backward.
SeenLastUse.insert(Op);
}
@@ -1922,9 +1939,7 @@ class RPFilter {
}
}
- DEBUG_SLSR_RP(dbgs() << "MaxWWithSLSR: " << MaxWWithSLSR << "\n");
-
- return MaxW;
+ return {MaxW, MaxWWithSLSR};
}
};
} // end of anonymous namespace
@@ -1951,8 +1966,7 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
if (Candidate *C = pickRewriteCandidate(I))
PickedCandidateMap[I] = C;
- // RPFilter RPFilter(&F, PickedCandidateMap, TTI);
- RPFilter RPFilter(&F, PickedCandidateMap);
+ RPFilter RPFilter(&F, PickedCandidateMap, TTI);
RPFilter.run();
#if 0
>From ee0a71e60f35e897b345104c1f06a807550aaf0f Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Tue, 25 Aug 2026 01:27:55 +0000
Subject: [PATCH 05/13] Add cache for regWeight
---
.../Scalar/StraightLineStrengthReduce.cpp | 33 +++++++++----------
1 file changed, 16 insertions(+), 17 deletions(-)
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 8fb8dced13072..20a309c0091a9 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -1635,11 +1635,23 @@ class RPFilter {
return isa<Instruction>(V) || isa<Argument>(V);
}
+ // The pressure scan asks for the weight of every live value at every
+ // instruction, so memoize on the type, which is all the weight depends on.
+ mutable DenseMap<Type *, unsigned> RegWeightCache;
+
unsigned regWeight(const Value *V) const {
- const DataLayout &DL = F->getDataLayout();
Type *Ty = V->getType();
if (Ty->isVoidTy() || Ty->isTokenTy())
return 0;
+
+ auto [It, Inserted] = RegWeightCache.try_emplace(Ty);
+ if (Inserted)
+ It->second = computeRegWeight(Ty);
+ return It->second;
+ }
+
+ unsigned computeRegWeight(Type *Ty) const {
+ const DataLayout &DL = F->getDataLayout();
if (UseTTIForRP) {
// Aggregates legalize to a flat sequence of scalars; approximate rather
// than calling getRegUsageForType, which is llvm_unreachable on them.
@@ -1659,11 +1671,10 @@ class RPFilter {
RegUsage *= SVT->getMinNumElements();
#endif
return RegUsage;
-
- } else {
- // Default logic to compute RP.
- return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
}
+
+ // Default logic to compute RP.
+ return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
}
void buildBBToLiveness(const Function &F) {
@@ -1918,14 +1929,12 @@ class RPFilter {
// Compute RP reaching for this instruction.
unsigned W = 0;
for (const Value *V : LiveSet) {
- // TODO: Cache regWeight(V)
auto RW = regWeight(V);
W += RW;
}
MaxW = std::max(MaxW, W);
W = 0;
for (const Value *V : LiveSetWithSLSR) {
- // TODO: Cache regWeight(V)
auto RW = regWeight(V);
W += RW;
}
@@ -1969,16 +1978,6 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
RPFilter RPFilter(&F, PickedCandidateMap, TTI);
RPFilter.run();
-#if 0
- DenseMap<const BasicBlock*, unsigned> BBToNumCands;
- for (auto &InstCand : PickedCandidateMap) {
- const BasicBlock *BB= InstCand.first->getParent();
- auto [It, Inserted] = BBToNumCands.try_emplace(BB);
- if (Inserted)
- It->second = countCandsInBB(It->first, PickedCandidateMap);
- }
-#endif
-
// From SortedCandidateInsts, remove some candidates that are likely to
// increase register pressure. The candidate's Inst is the source of
// replacement. A candidate in the following criteria should be removed:
>From 1b88d512ae9e6789b301d6f412ab3f5693614b86 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Tue, 25 Aug 2026 15:49:24 +0000
Subject: [PATCH 06/13] Add getRegisterBudget interface to TTI
---
.../llvm/Analysis/TargetTransformInfo.h | 11 ++
.../llvm/Analysis/TargetTransformInfoImpl.h | 5 +
llvm/lib/Analysis/TargetTransformInfo.cpp | 5 +
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 12 ++
.../Target/AMDGPU/AMDGPUTargetTransformInfo.h | 1 +
.../Scalar/StraightLineStrengthReduce.cpp | 104 +++++++++++++++++-
6 files changed, 132 insertions(+), 6 deletions(-)
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index e30cbc61a5420..e1d6c692f1c3b 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -1359,6 +1359,17 @@ class TargetTransformInfo {
/// \return the number of registers in the target-provided register class.
LLVM_ABI unsigned getNumberOfRegisters(unsigned ClassID) const;
+ /// \return The number of registers available to \p F before the register
+ /// allocator is forced to spill, or std::nullopt if the target cannot
+ /// provide a meaningful bound.
+ ///
+ /// Unlike getNumberOfRegisters(), which some targets deliberately
+ /// under-report to tune vectorization and interleaving, this is meant to be
+ /// a real budget usable by register-pressure heuristics. Targets with
+ /// several register files report the budget for the file that dominates
+ /// pressure.
+ LLVM_ABI std::optional<unsigned> getRegisterBudget(const Function &F) const;
+
/// \return true if the target supports load/store that enables fault
/// suppression of memory operands when the source condition is false.
LLVM_ABI bool hasConditionalLoadStoreForType(Type *Ty, bool IsStore) const;
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index 625433e2e0a0b..10794868855c7 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -614,6 +614,11 @@ class LLVM_ABI TargetTransformInfoImplBase {
}
virtual unsigned getNumberOfRegisters(unsigned ClassID) const { return 8; }
+
+ virtual std::optional<unsigned> getRegisterBudget(const Function &F) const {
+ return std::nullopt;
+ }
+
virtual bool hasConditionalLoadStoreForType(Type *Ty, bool IsStore) const {
return false;
}
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index 4c2cac9c440a0..a840803a096f6 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -823,6 +823,11 @@ unsigned TargetTransformInfo::getNumberOfRegisters(unsigned ClassID) const {
return TTIImpl->getNumberOfRegisters(ClassID);
}
+std::optional<unsigned>
+TargetTransformInfo::getRegisterBudget(const Function &F) const {
+ return TTIImpl->getRegisterBudget(F);
+}
+
bool TargetTransformInfo::hasConditionalLoadStoreForType(Type *Ty,
bool IsStore) const {
return TTIImpl->hasConditionalLoadStoreForType(Ty, IsStore);
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index ccb0c7314dcef..35a0d63e77f54 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -309,6 +309,18 @@ unsigned GCNTTIImpl::getNumberOfRegisters(unsigned RCID) const {
return 4;
}
+std::optional<unsigned> GCNTTIImpl::getRegisterBudget(const Function &F) const {
+ // Report the VGPR budget implied by the occupancy F is compiled for. Callers
+ // comparing a single lumped pressure number against this are conservative on
+ // the SGPR side, which is intentional: whether a value lands in an SGPR or a
+ // VGPR depends on divergence, not on its type.
+ //
+ // On GFX90A this is the combined VGPR+AGPR budget; see getMaxNumVectorRegs
+ // for the split. In dynamic VGPR mode "amdgpu-waves-per-eu" implies no VGPR
+ // limit at all, so this degrades to the full register file.
+ return ST->getMaxNumVGPRs(F);
+}
+
TypeSize
GCNTTIImpl::getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const {
switch (K) {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
index 4d9ff8d2d767f..d327046152991 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
@@ -128,6 +128,7 @@ class GCNTTIImpl final : public BasicTTIImplBase<GCNTTIImpl> {
}
unsigned getNumberOfRegisters(unsigned RCID) const override;
+ std::optional<unsigned> getRegisterBudget(const Function &F) const override;
TypeSize
getRegisterBitWidth(TargetTransformInfo::RegisterKind Vector) const override;
unsigned getMinVectorRegisterBitWidth() const override;
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 20a309c0091a9..6e8fb8eae1792 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -133,8 +133,31 @@ static cl::opt<int> SLSRBasisDistanceThreshold(
static cl::opt<bool> UseTTIForRP("slsr-tti-rp", cl::init(false),
cl::desc("Use TTI to compute RP for SLSR-RP"));
+static cl::opt<bool> EnableRPFilter(
+ "slsr-rp-filter", cl::init(false), cl::Hidden,
+ cl::desc("SLSR: skip rewrites in blocks where they would push register "
+ "pressure past the target's register budget"));
+
+static cl::opt<unsigned> SLSRRegBudget(
+ "slsr-reg-budget", cl::init(0), cl::Hidden,
+ cl::desc("SLSR: override the register budget reported by TTI"));
+
+static cl::opt<double> SLSRRPSafeFraction(
+ "slsr-rp-safe-fraction", cl::init(0.9), cl::Hidden,
+ cl::desc("SLSR: fraction of the register budget treated as safe"));
+
+static cl::opt<unsigned> SLSRRPAbsDelta(
+ "slsr-rp-abs-delta", cl::init(4), cl::Hidden,
+ cl::desc("SLSR: pressure increase tolerated in a block that is already "
+ "over the register budget"));
+
STATISTIC(NumSCEVCandidateBasisDifferences,
"Number of candidate-basis SCEV differences computed by SLSR");
+STATISTIC(NumRPFilteredBlocks,
+ "Number of blocks whose rewrites SLSR skipped due to register "
+ "pressure");
+STATISTIC(NumRPFilteredCandidates,
+ "Number of candidates SLSR skipped due to register pressure");
namespace {
@@ -1558,12 +1581,21 @@ class RPFilter {
const TargetTransformInfo *TTI)
: F(F), PickedCandidateMap(PickedCandidateMap), TTI(TTI) {}
- void run() {
+ // Candidates whose rewrite would push their block's register pressure past
+ // what the target can allocate are added to \p ToSkipRewrite.
+ void run(DenseSet<Instruction *> &ToSkipRewrite) {
buildBBToNumCandsAndBasises(PickedCandidateMap);
// TODO: 16 is an arbitrary threshold.
- if (MaxNumBasisesInBB > 16)
+ bool HaveLiveness = MaxNumBasisesInBB > 16;
+ if (HaveLiveness)
buildBBToLiveness(*F); // Do liveness analysis.
+ // The pressure numbers below only mean anything once liveness has been
+ // computed, and without a budget there is nothing to compare them to.
+ std::optional<unsigned> Budget;
+ if (EnableRPFilter && HaveLiveness)
+ Budget = getRegisterBudget();
+
DEBUG_SLSR_RP(dbgs() << "-- MaxRP of BBs -- \n");
for (auto &BB : *F) {
#if 0
@@ -1575,10 +1607,14 @@ class RPFilter {
dbgs() << "MaxRP:" << BB.getName() << ":" << RP << "," << RPBackward << "\n";
});
#else
- auto [MaxRP, MaxRPWithSLSR] = maxPressureInBlockBackward(
- BB, BBToLiveness[&BB].LiveIn, BBToLiveness[&BB].LiveOut);
+ const BlockLiveness &BL = getLiveness(&BB);
+ auto [MaxRP, MaxRPWithSLSR] =
+ maxPressureInBlockBackward(BB, BL.LiveIn, BL.LiveOut);
DEBUG_SLSR_RP(dbgs() << "MaxRP:" << BB.getName() << ": (" << MaxRP << ", "
<< MaxRPWithSLSR << ")" << "\n");
+
+ if (Budget && rewriteWouldOverflowBudget(MaxRP, MaxRPWithSLSR, *Budget))
+ skipRewritesInBlock(BB, ToSkipRewrite);
#endif
}
}
@@ -1600,6 +1636,60 @@ class RPFilter {
DenseMap<const BasicBlock *, BlockLiveness> BBToLiveness;
+ // Liveness is only computed for functions that pass the candidate-count
+ // gate. Blocks of the remaining functions read as having nothing live across
+ // their boundaries, which is why no rewrite is suppressed in that case.
+ const BlockLiveness &getLiveness(const BasicBlock *BB) const {
+ static const BlockLiveness Empty;
+ auto It = BBToLiveness.find(BB);
+ return It == BBToLiveness.end() ? Empty : It->second;
+ }
+
+ std::optional<unsigned> getRegisterBudget() const {
+ if (SLSRRegBudget > 0)
+ return SLSRRegBudget.getValue();
+ std::optional<unsigned> Budget = TTI->getRegisterBudget(*F);
+ if (Budget && *Budget == 0)
+ return std::nullopt;
+ return Budget;
+ }
+
+ // Return true if rewriting every candidate in a block, taking its peak
+ // pressure from \p Before to \p After, would ask the allocator for more
+ // registers than \p Budget.
+ bool rewriteWouldOverflowBudget(unsigned Before, unsigned After,
+ unsigned Budget) const {
+ // Leave the allocator some slack: it also has to satisfy register class
+ // and ABI constraints that this estimate knows nothing about.
+ unsigned SafeBudget = static_cast<unsigned>(Budget * SLSRRPSafeFraction);
+
+ // There is headroom, so however much the rewrite adds is irrelevant.
+ if (After <= SafeBudget)
+ return false;
+
+ // The rewrite is what takes the block over.
+ if (Before <= SafeBudget)
+ return true;
+
+ // Already over budget. SLSR can still lower pressure here, so only refuse
+ // rewrites that make it meaningfully worse.
+ return After > Before && After - Before > SLSRRPAbsDelta;
+ }
+
+ void skipRewritesInBlock(const BasicBlock &BB,
+ DenseSet<Instruction *> &ToSkipRewrite) {
+ ++NumRPFilteredBlocks;
+ DEBUG_SLSR_RP(dbgs() << "RP filter: skipping rewrites in " << BB.getName()
+ << "\n");
+ for (const Instruction &I : BB) {
+ auto It = PickedCandidateMap.find(&I);
+ if (It == PickedCandidateMap.end())
+ continue;
+ if (ToSkipRewrite.insert(It->first).second)
+ ++NumRPFilteredCandidates;
+ }
+ }
+
std::pair<unsigned, unsigned> countCandsAndBasisesInBB(
const BasicBlock *BB,
const DenseMap<Instruction *, Candidate *> &PickedCandidateMap) {
@@ -1975,8 +2065,11 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
if (Candidate *C = pickRewriteCandidate(I))
PickedCandidateMap[I] = C;
+ // Candidates whose rewrite is predicted to hurt more than it helps.
+ DenseSet<Instruction *> ToSkipRewrite;
+
RPFilter RPFilter(&F, PickedCandidateMap, TTI);
- RPFilter.run();
+ RPFilter.run(ToSkipRewrite);
// From SortedCandidateInsts, remove some candidates that are likely to
// increase register pressure. The candidate's Inst is the source of
@@ -2004,7 +2097,6 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
// Done before rewriting: rewriting inserts instructions and does
// replaceAllUsesWith, which would invalidate both the in-block index map and
// operands' user sets.
- DenseSet<Instruction *> ToSkipRewrite;
{
DenseMap<const BasicBlock *, DenseMap<const Instruction *, int>> IndexCache;
for (Instruction *I : SortedCandidateInsts)
>From 882874e7a174d56fbc1cd87892760da4da8257a6 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Tue, 1 Sep 2026 00:00:31 +0000
Subject: [PATCH 07/13] Clean up and adding a lit-test
---
.../Scalar/StraightLineStrengthReduce.cpp | 368 +---
.../AMDGPU/slsr-rp-filter.ll | 1922 +++++++++++++++++
.../slsr-basis-distance-threshold.ll | 94 -
.../slsr-rp-filter.ll | 296 +++
4 files changed, 2271 insertions(+), 409 deletions(-)
create mode 100644 llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
delete mode 100644 llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
create mode 100644 llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 6e8fb8eae1792..ae39823827087 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -125,16 +125,16 @@ static cl::opt<bool>
EnablePoisonReuseGuard("enable-poison-reuse-guard", cl::init(true),
cl::desc("Enable poison-reuse guard"));
-static cl::opt<int> SLSRBasisDistanceThreshold(
- "slsr-basis-distance-threshold", cl::init(96), cl::Hidden,
- cl::desc("SLSR: skip rewrite if in-block distance from Basis's last "
- "same-block use to the candidate Inst exceeds this"));
-
-static cl::opt<bool> UseTTIForRP("slsr-tti-rp", cl::init(false),
- cl::desc("Use TTI to compute RP for SLSR-RP"));
+// RPFilter targets one pathological shape: a block holding many distinct
+// bases. Each basis contributes one extended live range no matter how many
+// candidates are rewritten against it, so it is the number of distinct bases,
+// not the number of candidates, that tracks how many new concurrent live
+// ranges SLSR would create. Below this count no block in the function can
+// exhibit the pathology, and the liveness and pressure analyses are skipped.
+static constexpr unsigned MinDistinctBasesToFilter = 16;
static cl::opt<bool> EnableRPFilter(
- "slsr-rp-filter", cl::init(false), cl::Hidden,
+ "slsr-rp-filter", cl::init(true), cl::Hidden,
cl::desc("SLSR: skip rewrites in blocks where they would push register "
"pressure past the target's register budget"));
@@ -626,15 +626,6 @@ class StraightLineStrengthReduce {
if (auto *StrideInst = dyn_cast<Instruction>(C.Stride))
PropagateDependency(StrideInst);
};
-
- bool hasOperandsUsedInNonRewritableUsersInAnotherBlock(
- llvm::Instruction *Inst) const;
- bool hasRewritableCandidates(const Instruction *Inst) const;
- bool basisTooFarInSameBlock(
- const Candidate &C,
- DenseMap<const BasicBlock *, DenseMap<const Instruction *, int>>
- &IndexCache,
- const Instruction *Inst) const;
};
inline raw_ostream &operator<<(raw_ostream &OS,
@@ -1454,119 +1445,6 @@ bool StraightLineStrengthReduceLegacyPass::runOnFunction(Function &F) {
return StraightLineStrengthReduce(DL, DT, SE, TTI).runOnFunction(F);
}
-// Go through all operands of instruction, and check if any operand is used in
-// another block different from the instruction's block and the other use is
-// not rewritable. return true if such operand is found, otherwise return false.
-bool StraightLineStrengthReduce::
- hasOperandsUsedInNonRewritableUsersInAnotherBlock(
- llvm::Instruction *Inst) const {
- llvm::BasicBlock *InstBB = Inst->getParent();
-
- for (Value *OpVal : Inst->operand_values()) {
- auto *OpInst = dyn_cast<Instruction>(OpVal);
- if (!OpInst)
- continue;
-
- for (const User *U : OpInst->users())
- if (auto *UI = dyn_cast<Instruction>(U)) {
- if (UI->isDebugOrPseudoInst())
- continue;
- if (UI->getParent() != InstBB && !hasRewritableCandidates(UI)) {
- LLVM_DEBUG(dbgs()
- << "Inst's operand is used in another block "
- << "("
- << (InstBB->hasName() ? InstBB->getName() : "unnamed")
- << " -> "
- << (UI->getParent() && UI->getParent()->hasName()
- ? UI->getParent()->getName()
- : "unnamed")
- << ") " << *UI << "\n");
- return true;
- }
- }
- }
- return false;
-}
-
-bool StraightLineStrengthReduce::hasRewritableCandidates(
- const Instruction *Inst) const {
- if (!RewriteCandidates.count(Inst))
- return false;
-
- for (const Candidate *C : RewriteCandidates.at(Inst))
- if (C->Basis)
- return true;
-
- return false;
-}
-
-// Assign a monotonically increasing index to each (non-debug/pseudo)
-// instruction in BB so in-block distances can be queried in O(1) once built.
-static DenseMap<const Instruction *, int>
-buildBlockIndexMap(const BasicBlock &BB) {
- DenseMap<const Instruction *, int> IndexMap;
- int Index = 0;
- for (const Instruction &I : BB) {
- // Skip debug/pseudo instructions so the distance math is identical
- // between debug and release builds.
- if (I.isDebugOrPseudoInst())
- continue;
- IndexMap[&I] = Index++;
- }
- return IndexMap;
-}
-
-// Return true
-// 1. if C.Basis used only in C.Ins's block before C.Ins
-// AND
-// 2. if any use of C.Basis before C.Ins and C.Ins exceeds
-// SLSRBasisDistanceThreshold.
-bool StraightLineStrengthReduce::basisTooFarInSameBlock(
- const Candidate &C,
- DenseMap<const BasicBlock *, DenseMap<const Instruction *, int>>
- &IndexCache,
- const Instruction *I) const {
- Instruction *Inst = C.Ins;
- assert(Inst == I);
- Instruction *BasisInst = C.Basis ? C.Basis->Ins : nullptr;
- if (!BasisInst)
- return false;
-
- const BasicBlock *BB = Inst->getParent();
- auto [It, Inserted] = IndexCache.try_emplace(BB);
- if (Inserted)
- It->second = buildBlockIndexMap(*BB);
- const DenseMap<const Instruction *, int> &IndexMap = It->second;
-
- auto InstIt = IndexMap.find(Inst);
- if (InstIt == IndexMap.end())
- return false;
- int InstIdx = InstIt->second;
-
- int LastUseIdx = 0;
-
- bool FoundSameBlockUse = false;
- for (const User *U : BasisInst->users()) {
- const auto *UI = dyn_cast<Instruction>(U);
- if (!UI)
- continue;
- // If one of the uses is not in the same block, return false.
- if (UI->getParent() != BB)
- return false;
- auto UseIt = IndexMap.find(UI);
- // If any same block use is later than Inst, return false.
- if (UseIt == IndexMap.end() || UseIt->second >= InstIdx)
- return false;
- FoundSameBlockUse = true;
- LastUseIdx = std::max(LastUseIdx, UseIt->second);
- }
-
- if (!FoundSameBlockUse)
- return false;
-
- return (InstIdx - LastUseIdx) > SLSRBasisDistanceThreshold;
-}
-
namespace {
// TODO: Currently, I am considering (Basis, Cand) pair that are both in the
@@ -1581,42 +1459,53 @@ class RPFilter {
const TargetTransformInfo *TTI)
: F(F), PickedCandidateMap(PickedCandidateMap), TTI(TTI) {}
- // Candidates whose rewrite would push their block's register pressure past
- // what the target can allocate are added to \p ToSkipRewrite.
- void run(DenseSet<Instruction *> &ToSkipRewrite) {
+ // Return the candidates whose rewrite would push their block's register
+ // pressure past what the target can allocate.
+ DenseSet<const Instruction *> run() {
+ DenseSet<const Instruction *> InstsToSkip;
+ if (!EnableRPFilter || PickedCandidateMap.empty())
+ return InstsToSkip;
+
+ // Without a budget there is nothing to compare the pressure against, so
+ // check for one before paying for the liveness and pressure analyses.
+ std::optional<unsigned> Budget = getRegisterBudget();
+ if (!Budget)
+ return InstsToSkip;
+
buildBBToNumCandsAndBasises(PickedCandidateMap);
- // TODO: 16 is an arbitrary threshold.
- bool HaveLiveness = MaxNumBasisesInBB > 16;
- if (HaveLiveness)
- buildBBToLiveness(*F); // Do liveness analysis.
+ if (MaxNumBasisesInBB <= MinDistinctBasesToFilter)
+ return InstsToSkip;
- // The pressure numbers below only mean anything once liveness has been
- // computed, and without a budget there is nothing to compare them to.
- std::optional<unsigned> Budget;
- if (EnableRPFilter && HaveLiveness)
- Budget = getRegisterBudget();
+ // Compute live-in and live-out of each BB in CFG
+ buildBBToLiveness(*F);
DEBUG_SLSR_RP(dbgs() << "-- MaxRP of BBs -- \n");
+ SmallPtrSet<const BasicBlock *, 8> BBsToSkip;
for (auto &BB : *F) {
-#if 0
- unsigned RP = maxPressureInBlock(BB, BBToLiveness[&BB].LiveIn,
- BBToLiveness[&BB].LiveOut);
- unsigned RPBackward = maxPressureInBlockBackward(BB, BBToLiveness[&BB].LiveIn,
- BBToLiveness[&BB].LiveOut);
- DEBUG_WITH_TYPE("slsr-rp", {
- dbgs() << "MaxRP:" << BB.getName() << ":" << RP << "," << RPBackward << "\n";
- });
-#else
const BlockLiveness &BL = getLiveness(&BB);
auto [MaxRP, MaxRPWithSLSR] =
maxPressureInBlockBackward(BB, BL.LiveIn, BL.LiveOut);
DEBUG_SLSR_RP(dbgs() << "MaxRP:" << BB.getName() << ": (" << MaxRP << ", "
<< MaxRPWithSLSR << ")" << "\n");
- if (Budget && rewriteWouldOverflowBudget(MaxRP, MaxRPWithSLSR, *Budget))
- skipRewritesInBlock(BB, ToSkipRewrite);
-#endif
- }
+ if (!rewriteWouldOverflowBudget(MaxRP, MaxRPWithSLSR, *Budget))
+ continue;
+
+ DEBUG_SLSR_RP(dbgs() << "Skipping BB from SLSR: " << BB.getName()
+ << "\n");
+ ++NumRPFilteredBlocks;
+ BBsToSkip.insert(&BB);
+ } // Done with BBs
+
+ // One pass over the candidates rather than over the instructions of every
+ // skipped block. A skipped block is one of the larger blocks in the
+ // function, while the candidate map is small by comparison.
+ for (const auto &It : PickedCandidateMap)
+ if (BBsToSkip.contains(It.first->getParent()) &&
+ InstsToSkip.insert(It.first).second)
+ ++NumRPFilteredCandidates;
+
+ return InstsToSkip;
}
private:
@@ -1676,20 +1565,6 @@ class RPFilter {
return After > Before && After - Before > SLSRRPAbsDelta;
}
- void skipRewritesInBlock(const BasicBlock &BB,
- DenseSet<Instruction *> &ToSkipRewrite) {
- ++NumRPFilteredBlocks;
- DEBUG_SLSR_RP(dbgs() << "RP filter: skipping rewrites in " << BB.getName()
- << "\n");
- for (const Instruction &I : BB) {
- auto It = PickedCandidateMap.find(&I);
- if (It == PickedCandidateMap.end())
- continue;
- if (ToSkipRewrite.insert(It->first).second)
- ++NumRPFilteredCandidates;
- }
- }
-
std::pair<unsigned, unsigned> countCandsAndBasisesInBB(
const BasicBlock *BB,
const DenseMap<Instruction *, Candidate *> &PickedCandidateMap) {
@@ -1742,28 +1617,9 @@ class RPFilter {
unsigned computeRegWeight(Type *Ty) const {
const DataLayout &DL = F->getDataLayout();
- if (UseTTIForRP) {
- // Aggregates legalize to a flat sequence of scalars; approximate rather
- // than calling getRegUsageForType, which is llvm_unreachable on them.
- if (!VectorType::isValidElementType(Ty->getScalarType()))
- return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
-
- // TTI's getRegUsageForType can be used for types that are
- // validElementType(Ty->getScalarType()). However, even the valid vector
- // type should be multiplited by get{Min}NumElements() to handle
- // {ScalarVectorType}, FixedVectorType. Overall, getTypeSizeInBits(Ty)
- // handing all those cases should be sufficient for heuristic.
- unsigned RegUsage = TTI->getRegUsageForType(Ty);
-#if 0
- if (FixedVectorType *FVT = dyn_cast<FixedVectorType>(Ty))
- RegUsage *= FVT->getNumElements();
- else if (ScalableVectorType *SVT = dyn_cast<ScalableVectorType>(Ty))
- RegUsage *= SVT->getMinNumElements();
-#endif
- return RegUsage;
- }
- // Default logic to compute RP.
+ // TTI's getRegUsageForType is less accurate than
+ // default logic to compute RP for targets like AMDGPU
return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
}
@@ -1849,79 +1705,6 @@ class RPFilter {
return BB.getName() == "for.cond.cleanup";
}
- unsigned maxPressureInBlock(const BasicBlock &BB, const ValueSet &LiveIn,
- const ValueSet &LiveOut) const {
-
-#if 1
- bool IsDebugBlock = isDebugBlock(BB);
- unsigned StoreCount = 0;
-#endif
-
- SmallVector<const Instruction *, 128> Order;
- DenseMap<const Instruction *, unsigned> Idx;
- for (const Instruction &I : BB) {
- if (I.isDebugOrPseudoInst())
- continue;
- Idx[&I] = Order.size();
- Order.push_back(&I);
- }
-
- DenseMap<const Value *, unsigned> LastUse;
- for (const Instruction &I : BB) {
- if (I.isDebugOrPseudoInst() || isa<PHINode>(&I))
- continue;
- for (const Value *Op : I.operand_values())
- if (isRegisterLike(Op))
- LastUse[Op] = Idx[&I];
- }
- const unsigned End = Order.size();
- for (const Value *V : LiveOut)
- LastUse[V] = End; // survives the block; never retires here
-
- ValueSet Open;
- for (const Value *V : LiveIn)
- Open.insert(V);
-
- unsigned MaxW = 0;
- for (unsigned i = 0; i != End; ++i) {
- if (isa<StoreInst>(Order[i])) {
- StoreCount++;
- }
- if (!Order[i]->getType()->isVoidTy())
- Open.insert(Order[i]);
-
- unsigned W = 0;
- for (const Value *V : Open) {
- auto RW = regWeight(V);
- W += RW;
- if (IsDebugBlock && StoreCount == 32) {
- DEBUG_SLSR_RP({
- dbgs() << "regweight: " << "i: " << i << " value: " << *V
- << " RW: " << RW << ", ";
- });
- // dbgs() << "W: " << W << "\n";
- }
-
- // W += regWeight(V);
- }
- MaxW = std::max(MaxW, W);
-
- SmallVector<const Value *, 8> Dead;
- for (const Value *V : Open) {
- auto It = LastUse.find(V);
- if (It == LastUse.end() || It->second <= i)
- Dead.push_back(V);
- }
- for (const Value *V : Dead)
- Open.erase(V);
-
- if (IsDebugBlock && StoreCount == 32) {
- DEBUG_SLSR_RP(dbgs() << "W: " << W << " MaxW: " << MaxW << "\n");
- }
- }
- return MaxW;
- }
-
// Return true if I is a candidate and its basis is in the same bb, false
// otherwise.
bool insertBasisIfCand(const Instruction *I, ValueSet &LiveSetWithSLSR,
@@ -1943,17 +1726,6 @@ class RPFilter {
return false;
}
- // TODO: Remove
- void updateLiveSetWithSLSR(ValueSet &LiveSetWithSLSR,
- const DenseSet<const Value *> &SeenLastUse,
- const Instruction &I, const Value *Op) const {
- // LiveSet += {Cand.Basis} <-- Done already
- // LiveSet -= {Op} if this is the last use of Op (i.e.
- // SeenLastUse.contains(Op))
- if (SeenLastUse.contains(Op))
- LiveSetWithSLSR.erase(Op);
- }
-
std::pair<unsigned, unsigned>
maxPressureInBlockBackward(const BasicBlock &BB, const ValueSet &LiveIn,
const ValueSet &LiveOut) const {
@@ -2047,7 +1819,7 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
LLVM_DEBUG(dbgs() << "SLSR on Function: " << F.getName() << "\n");
// Traverse the dominator tree in the depth-first order. This order makes sure
// all bases of a candidate are in Candidates when we process it.
- for (auto *const Node : depth_first(DT))
+ for (const auto Node : depth_first(DT))
for (auto &I : *(Node->getBlock()))
allocateCandidatesAndFindBasis(&I);
@@ -2059,52 +1831,18 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
}
sortCandidateInstructions();
- ////////////////////////////////////////////////////
DenseMap<Instruction *, Candidate *> PickedCandidateMap;
for (Instruction *I : SortedCandidateInsts)
if (Candidate *C = pickRewriteCandidate(I))
PickedCandidateMap[I] = C;
- // Candidates whose rewrite is predicted to hurt more than it helps.
- DenseSet<Instruction *> ToSkipRewrite;
-
+ // Candidates whose rewrite would push their block's register pressure past
+ // what the target can allocate. Evaluated on the original IR, before any
+ // rewriteCandidate mutates it: rewriting inserts instructions and calls
+ // replaceAllUsesWith, which would invalidate the liveness and pressure
+ // analyses the filter relies on.
RPFilter RPFilter(&F, PickedCandidateMap, TTI);
- RPFilter.run(ToSkipRewrite);
-
- // From SortedCandidateInsts, remove some candidates that are likely to
- // increase register pressure. The candidate's Inst is the source of
- // replacement. A candidate in the following criteria should be removed:
- // 1. The candidate's Inst "has operands used in non-rewritable users in
- // another block"
- // -- checked by hasOperandsUsedInNonRewritableUsersInAnotherBlock(Inst)
- // -- This means the candidate's Inst's original operands are live in
- // another block, so even if rewrite the Inst, the operands may be still
- // live out to another block.
- // -- Thus, rewriting the Inst based on Basis might add another long live
- // range from the Basis by increasing the live range of the Basis.
- // -- TODO: If needed, a refinement to check that "another block" is
- // properly dominated by the candidate's Inst's block can be added.
- // 2. When the candidate's Basis's is only used in the the same block
- // and its last use is before the candidate's Inst, the difference between the
- // last use of Basis and the Inst is larger than a threshold.
- // -- This is also for avoiding increasing the live range of the Basis by
- // rewriting the Inst.
- //
- // A candidate satisfies both conditions 1 and 2 should be removed.
-
- // Collect candidates likely to increase register pressure.
- // Evaluate on the original IR, before any rewriteCandidate mutates it
- // Done before rewriting: rewriting inserts instructions and does
- // replaceAllUsesWith, which would invalidate both the in-block index map and
- // operands' user sets.
- {
- DenseMap<const BasicBlock *, DenseMap<const Instruction *, int>> IndexCache;
- for (Instruction *I : SortedCandidateInsts)
- if (Candidate *C = pickRewriteCandidate(I))
- if (hasOperandsUsedInNonRewritableUsersInAnotherBlock(I) &&
- basisTooFarInSameBlock(*C, IndexCache, I))
- ToSkipRewrite.insert(I);
- }
+ DenseSet<const Instruction *> ToSkipRewrite = RPFilter.run();
// Rewrite candidates in the topological order that rewrites a Candidate
// always before rewriting its Basis
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
new file mode 100644
index 0000000000000..24ae488d9a7f3
--- /dev/null
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
@@ -0,0 +1,1922 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -mtriple=amdgpu9.50-amd-amdhsa -passes=slsr -S -slsr-reg-budget=32 | FileCheck %s --check-prefixes=CHECK,BUDGET32
+; RUN: opt < %s -mtriple=amdgpu9.50-amd-amdhsa -passes=slsr -S | FileCheck %s --check-prefixes=CHECK,ATTRS
+; RUN: opt < %s -mtriple=amdgpu9.50-amd-amdhsa -passes=slsr -S -slsr-rp-filter=false | FileCheck %s --check-prefixes=CHECK,NOFILTER
+
+; SLSR rewrites a candidate as its basis plus a delta, which keeps the basis live
+; down to the candidate. The register-pressure filter drops a block's rewrites
+; when doing so would take the block's peak pressure past what the target can
+; allocate.
+;
+; The budget comes from TTI, which on AMDGPU reports the VGPR count implied by
+; the occupancy the function is compiled for, and -slsr-reg-budget overrides it.
+; -slsr-rp-safe-fraction (0.9 by default) is applied on top, so a budget of 32
+; leaves 28 usable registers. The filter does nothing unless -slsr-rp-filter is
+; passed, and it only examines functions where some block holds more than 16
+; distinct bases.
+;
+; @many_bases_overlapping has 17 bases and goes from 20 to 36 registers, crossing
+; the 28-register safe budget, so its rewrites are dropped. @many_bases_adjacent
+; has the same 17 bases, but each %u<k> immediately follows its %t<k>, so the
+; extended ranges never coexist and the peak only reaches 21, which fits.
+;
+; @four_bases holds only 4 bases, below the gate, so the filter never looks at it
+; and its rewrites stand. Its i128 values put the block at 28 registers rising to
+; 44, which would cross the 28-register safe budget if it were examined, so the
+; gate really is the only thing sparing it.
+;
+; The i32 functions carry no occupancy attributes, so their flat work group size
+; defaults to 1024, which is 16 wave64s spread over 4 EUs, hence 4 waves per EU
+; and a budget of 512/4 = 128 registers. Their peak of 36 fits, so the ATTRS run
+; leaves them alone and only the forced budget of 32 filters them.
+;
+; The i128 functions at the end raise the peak to 144 so the real budget decides
+; without any override. All three have identical bodies and identical pressure,
+; so the attributes are the only variable:
+;
+; @default_occupancy no attributes 4 waves/EU budget 128 filtered
+; @small_work_group flat-work-group-size 1 wave/EU budget 512 kept
+; @small_wg_high_occ + waves-per-eu=8 8 waves/EU budget 64 filtered
+;
+; The first pair shows flat-work-group-size raising the budget, and the second
+; shows waves-per-eu lowering it again. NOFILTER switches the analysis off.
+
+declare void @foo(i32)
+declare void @bar(i128)
+
+define void @many_bases_overlapping(i32 %s, i32 %b1, i32 %b2, i32 %b3, i32 %b4, i32 %b5, i32 %b6, i32 %b7, i32 %b8, i32 %b9, i32 %b10, i32 %b11, i32 %b12, i32 %b13, i32 %b14, i32 %b15, i32 %b16, i32 %b17) {
+; FILTER-LABEL: define void @many_bases_overlapping(
+; FILTER-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; FILTER-NEXT: [[ENTRY:.*:]]
+; FILTER-NEXT: [[S2:%.*]] = shl i32 [[S]], 1
+; FILTER-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T1]])
+; FILTER-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T2]])
+; FILTER-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T3]])
+; FILTER-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T4]])
+; FILTER-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T5]])
+; FILTER-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T6]])
+; FILTER-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T7]])
+; FILTER-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T8]])
+; FILTER-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T9]])
+; FILTER-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T10]])
+; FILTER-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T11]])
+; FILTER-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T12]])
+; FILTER-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T13]])
+; FILTER-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T14]])
+; FILTER-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T15]])
+; FILTER-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T16]])
+; FILTER-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
+; FILTER-NEXT: call void @foo(i32 [[T17]])
+; FILTER-NEXT: [[U1:%.*]] = add i32 [[B1]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U1]])
+; FILTER-NEXT: [[U2:%.*]] = add i32 [[B2]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U2]])
+; FILTER-NEXT: [[U3:%.*]] = add i32 [[B3]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U3]])
+; FILTER-NEXT: [[U4:%.*]] = add i32 [[B4]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U4]])
+; FILTER-NEXT: [[U5:%.*]] = add i32 [[B5]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U5]])
+; FILTER-NEXT: [[U6:%.*]] = add i32 [[B6]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U6]])
+; FILTER-NEXT: [[U7:%.*]] = add i32 [[B7]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U7]])
+; FILTER-NEXT: [[U8:%.*]] = add i32 [[B8]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U8]])
+; FILTER-NEXT: [[U9:%.*]] = add i32 [[B9]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U9]])
+; FILTER-NEXT: [[U10:%.*]] = add i32 [[B10]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U10]])
+; FILTER-NEXT: [[U11:%.*]] = add i32 [[B11]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U11]])
+; FILTER-NEXT: [[U12:%.*]] = add i32 [[B12]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U12]])
+; FILTER-NEXT: [[U13:%.*]] = add i32 [[B13]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U13]])
+; FILTER-NEXT: [[U14:%.*]] = add i32 [[B14]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U14]])
+; FILTER-NEXT: [[U15:%.*]] = add i32 [[B15]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U15]])
+; FILTER-NEXT: [[U16:%.*]] = add i32 [[B16]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U16]])
+; FILTER-NEXT: [[U17:%.*]] = add i32 [[B17]], [[S2]]
+; FILTER-NEXT: call void @foo(i32 [[U17]])
+; FILTER-NEXT: call void @foo(i32 [[B1]])
+; FILTER-NEXT: call void @foo(i32 [[B2]])
+; FILTER-NEXT: call void @foo(i32 [[B3]])
+; FILTER-NEXT: call void @foo(i32 [[B4]])
+; FILTER-NEXT: call void @foo(i32 [[B5]])
+; FILTER-NEXT: call void @foo(i32 [[B6]])
+; FILTER-NEXT: call void @foo(i32 [[B7]])
+; FILTER-NEXT: call void @foo(i32 [[B8]])
+; FILTER-NEXT: call void @foo(i32 [[B9]])
+; FILTER-NEXT: call void @foo(i32 [[B10]])
+; FILTER-NEXT: call void @foo(i32 [[B11]])
+; FILTER-NEXT: call void @foo(i32 [[B12]])
+; FILTER-NEXT: call void @foo(i32 [[B13]])
+; FILTER-NEXT: call void @foo(i32 [[B14]])
+; FILTER-NEXT: call void @foo(i32 [[B15]])
+; FILTER-NEXT: call void @foo(i32 [[B16]])
+; FILTER-NEXT: call void @foo(i32 [[B17]])
+; FILTER-NEXT: ret void
+;
+; OFF-LABEL: define void @many_bases_overlapping(
+; OFF-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; OFF-NEXT: [[ENTRY:.*:]]
+; OFF-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T1]])
+; OFF-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T2]])
+; OFF-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T3]])
+; OFF-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T4]])
+; OFF-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T5]])
+; OFF-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T6]])
+; OFF-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T7]])
+; OFF-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T8]])
+; OFF-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T9]])
+; OFF-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T10]])
+; OFF-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T11]])
+; OFF-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T12]])
+; OFF-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T13]])
+; OFF-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T14]])
+; OFF-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T15]])
+; OFF-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T16]])
+; OFF-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[T17]])
+; OFF-NEXT: [[U1:%.*]] = add i32 [[T1]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U1]])
+; OFF-NEXT: [[U2:%.*]] = add i32 [[T2]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U2]])
+; OFF-NEXT: [[U3:%.*]] = add i32 [[T3]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U3]])
+; OFF-NEXT: [[U4:%.*]] = add i32 [[T4]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U4]])
+; OFF-NEXT: [[U5:%.*]] = add i32 [[T5]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U5]])
+; OFF-NEXT: [[U6:%.*]] = add i32 [[T6]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U6]])
+; OFF-NEXT: [[U7:%.*]] = add i32 [[T7]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U7]])
+; OFF-NEXT: [[U8:%.*]] = add i32 [[T8]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U8]])
+; OFF-NEXT: [[U9:%.*]] = add i32 [[T9]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U9]])
+; OFF-NEXT: [[U10:%.*]] = add i32 [[T10]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U10]])
+; OFF-NEXT: [[U11:%.*]] = add i32 [[T11]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U11]])
+; OFF-NEXT: [[U12:%.*]] = add i32 [[T12]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U12]])
+; OFF-NEXT: [[U13:%.*]] = add i32 [[T13]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U13]])
+; OFF-NEXT: [[U14:%.*]] = add i32 [[T14]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U14]])
+; OFF-NEXT: [[U15:%.*]] = add i32 [[T15]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U15]])
+; OFF-NEXT: [[U16:%.*]] = add i32 [[T16]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U16]])
+; OFF-NEXT: [[U17:%.*]] = add i32 [[T17]], [[S]]
+; OFF-NEXT: call void @foo(i32 [[U17]])
+; OFF-NEXT: call void @foo(i32 [[B1]])
+; OFF-NEXT: call void @foo(i32 [[B2]])
+; OFF-NEXT: call void @foo(i32 [[B3]])
+; OFF-NEXT: call void @foo(i32 [[B4]])
+; OFF-NEXT: call void @foo(i32 [[B5]])
+; OFF-NEXT: call void @foo(i32 [[B6]])
+; OFF-NEXT: call void @foo(i32 [[B7]])
+; OFF-NEXT: call void @foo(i32 [[B8]])
+; OFF-NEXT: call void @foo(i32 [[B9]])
+; OFF-NEXT: call void @foo(i32 [[B10]])
+; OFF-NEXT: call void @foo(i32 [[B11]])
+; OFF-NEXT: call void @foo(i32 [[B12]])
+; OFF-NEXT: call void @foo(i32 [[B13]])
+; OFF-NEXT: call void @foo(i32 [[B14]])
+; OFF-NEXT: call void @foo(i32 [[B15]])
+; OFF-NEXT: call void @foo(i32 [[B16]])
+; OFF-NEXT: call void @foo(i32 [[B17]])
+; OFF-NEXT: ret void
+;
+; BUDGET32-LABEL: define void @many_bases_overlapping(
+; BUDGET32-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; BUDGET32-NEXT: [[ENTRY:.*:]]
+; BUDGET32-NEXT: [[S2:%.*]] = shl i32 [[S]], 1
+; BUDGET32-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T1]])
+; BUDGET32-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T2]])
+; BUDGET32-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T3]])
+; BUDGET32-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T4]])
+; BUDGET32-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T5]])
+; BUDGET32-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T6]])
+; BUDGET32-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T7]])
+; BUDGET32-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T8]])
+; BUDGET32-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T9]])
+; BUDGET32-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T10]])
+; BUDGET32-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T11]])
+; BUDGET32-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T12]])
+; BUDGET32-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T13]])
+; BUDGET32-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T14]])
+; BUDGET32-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T15]])
+; BUDGET32-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T16]])
+; BUDGET32-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
+; BUDGET32-NEXT: call void @foo(i32 [[T17]])
+; BUDGET32-NEXT: [[U1:%.*]] = add i32 [[B1]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U1]])
+; BUDGET32-NEXT: [[U2:%.*]] = add i32 [[B2]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U2]])
+; BUDGET32-NEXT: [[U3:%.*]] = add i32 [[B3]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U3]])
+; BUDGET32-NEXT: [[U4:%.*]] = add i32 [[B4]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U4]])
+; BUDGET32-NEXT: [[U5:%.*]] = add i32 [[B5]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U5]])
+; BUDGET32-NEXT: [[U6:%.*]] = add i32 [[B6]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U6]])
+; BUDGET32-NEXT: [[U7:%.*]] = add i32 [[B7]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U7]])
+; BUDGET32-NEXT: [[U8:%.*]] = add i32 [[B8]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U8]])
+; BUDGET32-NEXT: [[U9:%.*]] = add i32 [[B9]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U9]])
+; BUDGET32-NEXT: [[U10:%.*]] = add i32 [[B10]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U10]])
+; BUDGET32-NEXT: [[U11:%.*]] = add i32 [[B11]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U11]])
+; BUDGET32-NEXT: [[U12:%.*]] = add i32 [[B12]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U12]])
+; BUDGET32-NEXT: [[U13:%.*]] = add i32 [[B13]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U13]])
+; BUDGET32-NEXT: [[U14:%.*]] = add i32 [[B14]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U14]])
+; BUDGET32-NEXT: [[U15:%.*]] = add i32 [[B15]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U15]])
+; BUDGET32-NEXT: [[U16:%.*]] = add i32 [[B16]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U16]])
+; BUDGET32-NEXT: [[U17:%.*]] = add i32 [[B17]], [[S2]]
+; BUDGET32-NEXT: call void @foo(i32 [[U17]])
+; BUDGET32-NEXT: call void @foo(i32 [[B1]])
+; BUDGET32-NEXT: call void @foo(i32 [[B2]])
+; BUDGET32-NEXT: call void @foo(i32 [[B3]])
+; BUDGET32-NEXT: call void @foo(i32 [[B4]])
+; BUDGET32-NEXT: call void @foo(i32 [[B5]])
+; BUDGET32-NEXT: call void @foo(i32 [[B6]])
+; BUDGET32-NEXT: call void @foo(i32 [[B7]])
+; BUDGET32-NEXT: call void @foo(i32 [[B8]])
+; BUDGET32-NEXT: call void @foo(i32 [[B9]])
+; BUDGET32-NEXT: call void @foo(i32 [[B10]])
+; BUDGET32-NEXT: call void @foo(i32 [[B11]])
+; BUDGET32-NEXT: call void @foo(i32 [[B12]])
+; BUDGET32-NEXT: call void @foo(i32 [[B13]])
+; BUDGET32-NEXT: call void @foo(i32 [[B14]])
+; BUDGET32-NEXT: call void @foo(i32 [[B15]])
+; BUDGET32-NEXT: call void @foo(i32 [[B16]])
+; BUDGET32-NEXT: call void @foo(i32 [[B17]])
+; BUDGET32-NEXT: ret void
+;
+; ATTRS-LABEL: define void @many_bases_overlapping(
+; ATTRS-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; ATTRS-NEXT: [[ENTRY:.*:]]
+; ATTRS-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T1]])
+; ATTRS-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T2]])
+; ATTRS-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T3]])
+; ATTRS-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T4]])
+; ATTRS-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T5]])
+; ATTRS-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T6]])
+; ATTRS-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T7]])
+; ATTRS-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T8]])
+; ATTRS-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T9]])
+; ATTRS-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T10]])
+; ATTRS-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T11]])
+; ATTRS-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T12]])
+; ATTRS-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T13]])
+; ATTRS-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T14]])
+; ATTRS-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T15]])
+; ATTRS-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T16]])
+; ATTRS-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[T17]])
+; ATTRS-NEXT: [[U1:%.*]] = add i32 [[T1]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U1]])
+; ATTRS-NEXT: [[U2:%.*]] = add i32 [[T2]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U2]])
+; ATTRS-NEXT: [[U3:%.*]] = add i32 [[T3]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U3]])
+; ATTRS-NEXT: [[U4:%.*]] = add i32 [[T4]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U4]])
+; ATTRS-NEXT: [[U5:%.*]] = add i32 [[T5]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U5]])
+; ATTRS-NEXT: [[U6:%.*]] = add i32 [[T6]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U6]])
+; ATTRS-NEXT: [[U7:%.*]] = add i32 [[T7]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U7]])
+; ATTRS-NEXT: [[U8:%.*]] = add i32 [[T8]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U8]])
+; ATTRS-NEXT: [[U9:%.*]] = add i32 [[T9]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U9]])
+; ATTRS-NEXT: [[U10:%.*]] = add i32 [[T10]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U10]])
+; ATTRS-NEXT: [[U11:%.*]] = add i32 [[T11]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U11]])
+; ATTRS-NEXT: [[U12:%.*]] = add i32 [[T12]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U12]])
+; ATTRS-NEXT: [[U13:%.*]] = add i32 [[T13]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U13]])
+; ATTRS-NEXT: [[U14:%.*]] = add i32 [[T14]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U14]])
+; ATTRS-NEXT: [[U15:%.*]] = add i32 [[T15]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U15]])
+; ATTRS-NEXT: [[U16:%.*]] = add i32 [[T16]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U16]])
+; ATTRS-NEXT: [[U17:%.*]] = add i32 [[T17]], [[S]]
+; ATTRS-NEXT: call void @foo(i32 [[U17]])
+; ATTRS-NEXT: call void @foo(i32 [[B1]])
+; ATTRS-NEXT: call void @foo(i32 [[B2]])
+; ATTRS-NEXT: call void @foo(i32 [[B3]])
+; ATTRS-NEXT: call void @foo(i32 [[B4]])
+; ATTRS-NEXT: call void @foo(i32 [[B5]])
+; ATTRS-NEXT: call void @foo(i32 [[B6]])
+; ATTRS-NEXT: call void @foo(i32 [[B7]])
+; ATTRS-NEXT: call void @foo(i32 [[B8]])
+; ATTRS-NEXT: call void @foo(i32 [[B9]])
+; ATTRS-NEXT: call void @foo(i32 [[B10]])
+; ATTRS-NEXT: call void @foo(i32 [[B11]])
+; ATTRS-NEXT: call void @foo(i32 [[B12]])
+; ATTRS-NEXT: call void @foo(i32 [[B13]])
+; ATTRS-NEXT: call void @foo(i32 [[B14]])
+; ATTRS-NEXT: call void @foo(i32 [[B15]])
+; ATTRS-NEXT: call void @foo(i32 [[B16]])
+; ATTRS-NEXT: call void @foo(i32 [[B17]])
+; ATTRS-NEXT: ret void
+;
+; NOFILTER-LABEL: define void @many_bases_overlapping(
+; NOFILTER-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; NOFILTER-NEXT: [[ENTRY:.*:]]
+; NOFILTER-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T1]])
+; NOFILTER-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T2]])
+; NOFILTER-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T3]])
+; NOFILTER-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T4]])
+; NOFILTER-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T5]])
+; NOFILTER-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T6]])
+; NOFILTER-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T7]])
+; NOFILTER-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T8]])
+; NOFILTER-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T9]])
+; NOFILTER-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T10]])
+; NOFILTER-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T11]])
+; NOFILTER-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T12]])
+; NOFILTER-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T13]])
+; NOFILTER-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T14]])
+; NOFILTER-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T15]])
+; NOFILTER-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T16]])
+; NOFILTER-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[T17]])
+; NOFILTER-NEXT: [[U1:%.*]] = add i32 [[T1]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U1]])
+; NOFILTER-NEXT: [[U2:%.*]] = add i32 [[T2]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U2]])
+; NOFILTER-NEXT: [[U3:%.*]] = add i32 [[T3]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U3]])
+; NOFILTER-NEXT: [[U4:%.*]] = add i32 [[T4]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U4]])
+; NOFILTER-NEXT: [[U5:%.*]] = add i32 [[T5]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U5]])
+; NOFILTER-NEXT: [[U6:%.*]] = add i32 [[T6]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U6]])
+; NOFILTER-NEXT: [[U7:%.*]] = add i32 [[T7]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U7]])
+; NOFILTER-NEXT: [[U8:%.*]] = add i32 [[T8]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U8]])
+; NOFILTER-NEXT: [[U9:%.*]] = add i32 [[T9]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U9]])
+; NOFILTER-NEXT: [[U10:%.*]] = add i32 [[T10]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U10]])
+; NOFILTER-NEXT: [[U11:%.*]] = add i32 [[T11]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U11]])
+; NOFILTER-NEXT: [[U12:%.*]] = add i32 [[T12]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U12]])
+; NOFILTER-NEXT: [[U13:%.*]] = add i32 [[T13]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U13]])
+; NOFILTER-NEXT: [[U14:%.*]] = add i32 [[T14]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U14]])
+; NOFILTER-NEXT: [[U15:%.*]] = add i32 [[T15]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U15]])
+; NOFILTER-NEXT: [[U16:%.*]] = add i32 [[T16]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U16]])
+; NOFILTER-NEXT: [[U17:%.*]] = add i32 [[T17]], [[S]]
+; NOFILTER-NEXT: call void @foo(i32 [[U17]])
+; NOFILTER-NEXT: call void @foo(i32 [[B1]])
+; NOFILTER-NEXT: call void @foo(i32 [[B2]])
+; NOFILTER-NEXT: call void @foo(i32 [[B3]])
+; NOFILTER-NEXT: call void @foo(i32 [[B4]])
+; NOFILTER-NEXT: call void @foo(i32 [[B5]])
+; NOFILTER-NEXT: call void @foo(i32 [[B6]])
+; NOFILTER-NEXT: call void @foo(i32 [[B7]])
+; NOFILTER-NEXT: call void @foo(i32 [[B8]])
+; NOFILTER-NEXT: call void @foo(i32 [[B9]])
+; NOFILTER-NEXT: call void @foo(i32 [[B10]])
+; NOFILTER-NEXT: call void @foo(i32 [[B11]])
+; NOFILTER-NEXT: call void @foo(i32 [[B12]])
+; NOFILTER-NEXT: call void @foo(i32 [[B13]])
+; NOFILTER-NEXT: call void @foo(i32 [[B14]])
+; NOFILTER-NEXT: call void @foo(i32 [[B15]])
+; NOFILTER-NEXT: call void @foo(i32 [[B16]])
+; NOFILTER-NEXT: call void @foo(i32 [[B17]])
+; NOFILTER-NEXT: ret void
+;
+entry:
+ %s2 = shl i32 %s, 1
+ %t1 = add i32 %b1, %s
+ call void @foo(i32 %t1)
+ %t2 = add i32 %b2, %s
+ call void @foo(i32 %t2)
+ %t3 = add i32 %b3, %s
+ call void @foo(i32 %t3)
+ %t4 = add i32 %b4, %s
+ call void @foo(i32 %t4)
+ %t5 = add i32 %b5, %s
+ call void @foo(i32 %t5)
+ %t6 = add i32 %b6, %s
+ call void @foo(i32 %t6)
+ %t7 = add i32 %b7, %s
+ call void @foo(i32 %t7)
+ %t8 = add i32 %b8, %s
+ call void @foo(i32 %t8)
+ %t9 = add i32 %b9, %s
+ call void @foo(i32 %t9)
+ %t10 = add i32 %b10, %s
+ call void @foo(i32 %t10)
+ %t11 = add i32 %b11, %s
+ call void @foo(i32 %t11)
+ %t12 = add i32 %b12, %s
+ call void @foo(i32 %t12)
+ %t13 = add i32 %b13, %s
+ call void @foo(i32 %t13)
+ %t14 = add i32 %b14, %s
+ call void @foo(i32 %t14)
+ %t15 = add i32 %b15, %s
+ call void @foo(i32 %t15)
+ %t16 = add i32 %b16, %s
+ call void @foo(i32 %t16)
+ %t17 = add i32 %b17, %s
+ call void @foo(i32 %t17)
+ %u1 = add i32 %b1, %s2
+ call void @foo(i32 %u1)
+ %u2 = add i32 %b2, %s2
+ call void @foo(i32 %u2)
+ %u3 = add i32 %b3, %s2
+ call void @foo(i32 %u3)
+ %u4 = add i32 %b4, %s2
+ call void @foo(i32 %u4)
+ %u5 = add i32 %b5, %s2
+ call void @foo(i32 %u5)
+ %u6 = add i32 %b6, %s2
+ call void @foo(i32 %u6)
+ %u7 = add i32 %b7, %s2
+ call void @foo(i32 %u7)
+ %u8 = add i32 %b8, %s2
+ call void @foo(i32 %u8)
+ %u9 = add i32 %b9, %s2
+ call void @foo(i32 %u9)
+ %u10 = add i32 %b10, %s2
+ call void @foo(i32 %u10)
+ %u11 = add i32 %b11, %s2
+ call void @foo(i32 %u11)
+ %u12 = add i32 %b12, %s2
+ call void @foo(i32 %u12)
+ %u13 = add i32 %b13, %s2
+ call void @foo(i32 %u13)
+ %u14 = add i32 %b14, %s2
+ call void @foo(i32 %u14)
+ %u15 = add i32 %b15, %s2
+ call void @foo(i32 %u15)
+ %u16 = add i32 %b16, %s2
+ call void @foo(i32 %u16)
+ %u17 = add i32 %b17, %s2
+ call void @foo(i32 %u17)
+ call void @foo(i32 %b1)
+ call void @foo(i32 %b2)
+ call void @foo(i32 %b3)
+ call void @foo(i32 %b4)
+ call void @foo(i32 %b5)
+ call void @foo(i32 %b6)
+ call void @foo(i32 %b7)
+ call void @foo(i32 %b8)
+ call void @foo(i32 %b9)
+ call void @foo(i32 %b10)
+ call void @foo(i32 %b11)
+ call void @foo(i32 %b12)
+ call void @foo(i32 %b13)
+ call void @foo(i32 %b14)
+ call void @foo(i32 %b15)
+ call void @foo(i32 %b16)
+ call void @foo(i32 %b17)
+ ret void
+}
+
+define void @many_bases_adjacent(i32 %s, i32 %b1, i32 %b2, i32 %b3, i32 %b4, i32 %b5, i32 %b6, i32 %b7, i32 %b8, i32 %b9, i32 %b10, i32 %b11, i32 %b12, i32 %b13, i32 %b14, i32 %b15, i32 %b16, i32 %b17) {
+; CHECK-LABEL: define void @many_bases_adjacent(
+; CHECK-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T1]])
+; CHECK-NEXT: [[U1:%.*]] = add i32 [[T1]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U1]])
+; CHECK-NEXT: call void @foo(i32 [[B1]])
+; CHECK-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T2]])
+; CHECK-NEXT: [[U2:%.*]] = add i32 [[T2]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U2]])
+; CHECK-NEXT: call void @foo(i32 [[B2]])
+; CHECK-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T3]])
+; CHECK-NEXT: [[U3:%.*]] = add i32 [[T3]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U3]])
+; CHECK-NEXT: call void @foo(i32 [[B3]])
+; CHECK-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T4]])
+; CHECK-NEXT: [[U4:%.*]] = add i32 [[T4]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U4]])
+; CHECK-NEXT: call void @foo(i32 [[B4]])
+; CHECK-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T5]])
+; CHECK-NEXT: [[U5:%.*]] = add i32 [[T5]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U5]])
+; CHECK-NEXT: call void @foo(i32 [[B5]])
+; CHECK-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T6]])
+; CHECK-NEXT: [[U6:%.*]] = add i32 [[T6]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U6]])
+; CHECK-NEXT: call void @foo(i32 [[B6]])
+; CHECK-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T7]])
+; CHECK-NEXT: [[U7:%.*]] = add i32 [[T7]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U7]])
+; CHECK-NEXT: call void @foo(i32 [[B7]])
+; CHECK-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T8]])
+; CHECK-NEXT: [[U8:%.*]] = add i32 [[T8]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U8]])
+; CHECK-NEXT: call void @foo(i32 [[B8]])
+; CHECK-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T9]])
+; CHECK-NEXT: [[U9:%.*]] = add i32 [[T9]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U9]])
+; CHECK-NEXT: call void @foo(i32 [[B9]])
+; CHECK-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T10]])
+; CHECK-NEXT: [[U10:%.*]] = add i32 [[T10]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U10]])
+; CHECK-NEXT: call void @foo(i32 [[B10]])
+; CHECK-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T11]])
+; CHECK-NEXT: [[U11:%.*]] = add i32 [[T11]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U11]])
+; CHECK-NEXT: call void @foo(i32 [[B11]])
+; CHECK-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T12]])
+; CHECK-NEXT: [[U12:%.*]] = add i32 [[T12]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U12]])
+; CHECK-NEXT: call void @foo(i32 [[B12]])
+; CHECK-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T13]])
+; CHECK-NEXT: [[U13:%.*]] = add i32 [[T13]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U13]])
+; CHECK-NEXT: call void @foo(i32 [[B13]])
+; CHECK-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T14]])
+; CHECK-NEXT: [[U14:%.*]] = add i32 [[T14]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U14]])
+; CHECK-NEXT: call void @foo(i32 [[B14]])
+; CHECK-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T15]])
+; CHECK-NEXT: [[U15:%.*]] = add i32 [[T15]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U15]])
+; CHECK-NEXT: call void @foo(i32 [[B15]])
+; CHECK-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T16]])
+; CHECK-NEXT: [[U16:%.*]] = add i32 [[T16]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U16]])
+; CHECK-NEXT: call void @foo(i32 [[B16]])
+; CHECK-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[T17]])
+; CHECK-NEXT: [[U17:%.*]] = add i32 [[T17]], [[S]]
+; CHECK-NEXT: call void @foo(i32 [[U17]])
+; CHECK-NEXT: call void @foo(i32 [[B17]])
+; CHECK-NEXT: ret void
+;
+entry:
+ %s2 = shl i32 %s, 1
+ %t1 = add i32 %b1, %s
+ call void @foo(i32 %t1)
+ %u1 = add i32 %b1, %s2
+ call void @foo(i32 %u1)
+ call void @foo(i32 %b1)
+ %t2 = add i32 %b2, %s
+ call void @foo(i32 %t2)
+ %u2 = add i32 %b2, %s2
+ call void @foo(i32 %u2)
+ call void @foo(i32 %b2)
+ %t3 = add i32 %b3, %s
+ call void @foo(i32 %t3)
+ %u3 = add i32 %b3, %s2
+ call void @foo(i32 %u3)
+ call void @foo(i32 %b3)
+ %t4 = add i32 %b4, %s
+ call void @foo(i32 %t4)
+ %u4 = add i32 %b4, %s2
+ call void @foo(i32 %u4)
+ call void @foo(i32 %b4)
+ %t5 = add i32 %b5, %s
+ call void @foo(i32 %t5)
+ %u5 = add i32 %b5, %s2
+ call void @foo(i32 %u5)
+ call void @foo(i32 %b5)
+ %t6 = add i32 %b6, %s
+ call void @foo(i32 %t6)
+ %u6 = add i32 %b6, %s2
+ call void @foo(i32 %u6)
+ call void @foo(i32 %b6)
+ %t7 = add i32 %b7, %s
+ call void @foo(i32 %t7)
+ %u7 = add i32 %b7, %s2
+ call void @foo(i32 %u7)
+ call void @foo(i32 %b7)
+ %t8 = add i32 %b8, %s
+ call void @foo(i32 %t8)
+ %u8 = add i32 %b8, %s2
+ call void @foo(i32 %u8)
+ call void @foo(i32 %b8)
+ %t9 = add i32 %b9, %s
+ call void @foo(i32 %t9)
+ %u9 = add i32 %b9, %s2
+ call void @foo(i32 %u9)
+ call void @foo(i32 %b9)
+ %t10 = add i32 %b10, %s
+ call void @foo(i32 %t10)
+ %u10 = add i32 %b10, %s2
+ call void @foo(i32 %u10)
+ call void @foo(i32 %b10)
+ %t11 = add i32 %b11, %s
+ call void @foo(i32 %t11)
+ %u11 = add i32 %b11, %s2
+ call void @foo(i32 %u11)
+ call void @foo(i32 %b11)
+ %t12 = add i32 %b12, %s
+ call void @foo(i32 %t12)
+ %u12 = add i32 %b12, %s2
+ call void @foo(i32 %u12)
+ call void @foo(i32 %b12)
+ %t13 = add i32 %b13, %s
+ call void @foo(i32 %t13)
+ %u13 = add i32 %b13, %s2
+ call void @foo(i32 %u13)
+ call void @foo(i32 %b13)
+ %t14 = add i32 %b14, %s
+ call void @foo(i32 %t14)
+ %u14 = add i32 %b14, %s2
+ call void @foo(i32 %u14)
+ call void @foo(i32 %b14)
+ %t15 = add i32 %b15, %s
+ call void @foo(i32 %t15)
+ %u15 = add i32 %b15, %s2
+ call void @foo(i32 %u15)
+ call void @foo(i32 %b15)
+ %t16 = add i32 %b16, %s
+ call void @foo(i32 %t16)
+ %u16 = add i32 %b16, %s2
+ call void @foo(i32 %u16)
+ call void @foo(i32 %b16)
+ %t17 = add i32 %b17, %s
+ call void @foo(i32 %t17)
+ %u17 = add i32 %b17, %s2
+ call void @foo(i32 %u17)
+ call void @foo(i32 %b17)
+ ret void
+}
+
+define void @four_bases(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4) {
+; CHECK-LABEL: define void @four_bases(
+; CHECK-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T1]])
+; CHECK-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T2]])
+; CHECK-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T3]])
+; CHECK-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T4]])
+; CHECK-NEXT: [[U1:%.*]] = add i128 [[T1]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U1]])
+; CHECK-NEXT: [[U2:%.*]] = add i128 [[T2]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U2]])
+; CHECK-NEXT: [[U3:%.*]] = add i128 [[T3]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U3]])
+; CHECK-NEXT: [[U4:%.*]] = add i128 [[T4]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U4]])
+; CHECK-NEXT: call void @bar(i128 [[B1]])
+; CHECK-NEXT: call void @bar(i128 [[B2]])
+; CHECK-NEXT: call void @bar(i128 [[B3]])
+; CHECK-NEXT: call void @bar(i128 [[B4]])
+; CHECK-NEXT: ret void
+;
+entry:
+ %s2 = shl i128 %s, 1
+ %t1 = add i128 %b1, %s
+ call void @bar(i128 %t1)
+ %t2 = add i128 %b2, %s
+ call void @bar(i128 %t2)
+ %t3 = add i128 %b3, %s
+ call void @bar(i128 %t3)
+ %t4 = add i128 %b4, %s
+ call void @bar(i128 %t4)
+ %u1 = add i128 %b1, %s2
+ call void @bar(i128 %u1)
+ %u2 = add i128 %b2, %s2
+ call void @bar(i128 %u2)
+ %u3 = add i128 %b3, %s2
+ call void @bar(i128 %u3)
+ %u4 = add i128 %b4, %s2
+ call void @bar(i128 %u4)
+ call void @bar(i128 %b1)
+ call void @bar(i128 %b2)
+ call void @bar(i128 %b3)
+ call void @bar(i128 %b4)
+ ret void
+}
+
+; i128 values weigh 4 registers each, so these three bodies peak at 144 with the
+; rewrites applied against 80 without. That straddles the budgets implied by the
+; occupancy attributes, letting the real TTI budget decide with no override.
+
+; No attributes: flat work group size defaults to 1024, so 4 waves per EU and a
+; 128-register budget. 144 exceeds the 115-register safe budget, so filtered.
+define void @default_occupancy(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) {
+; BUDGET32-LABEL: define void @default_occupancy(
+; BUDGET32-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) {
+; BUDGET32-NEXT: [[ENTRY:.*:]]
+; BUDGET32-NEXT: [[S2:%.*]] = shl i128 [[S]], 1
+; BUDGET32-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T1]])
+; BUDGET32-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T2]])
+; BUDGET32-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T3]])
+; BUDGET32-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T4]])
+; BUDGET32-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T5]])
+; BUDGET32-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T6]])
+; BUDGET32-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T7]])
+; BUDGET32-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T8]])
+; BUDGET32-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T9]])
+; BUDGET32-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T10]])
+; BUDGET32-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T11]])
+; BUDGET32-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T12]])
+; BUDGET32-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T13]])
+; BUDGET32-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T14]])
+; BUDGET32-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T15]])
+; BUDGET32-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T16]])
+; BUDGET32-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T17]])
+; BUDGET32-NEXT: [[U1:%.*]] = add i128 [[B1]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U1]])
+; BUDGET32-NEXT: [[U2:%.*]] = add i128 [[B2]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U2]])
+; BUDGET32-NEXT: [[U3:%.*]] = add i128 [[B3]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U3]])
+; BUDGET32-NEXT: [[U4:%.*]] = add i128 [[B4]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U4]])
+; BUDGET32-NEXT: [[U5:%.*]] = add i128 [[B5]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U5]])
+; BUDGET32-NEXT: [[U6:%.*]] = add i128 [[B6]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U6]])
+; BUDGET32-NEXT: [[U7:%.*]] = add i128 [[B7]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U7]])
+; BUDGET32-NEXT: [[U8:%.*]] = add i128 [[B8]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U8]])
+; BUDGET32-NEXT: [[U9:%.*]] = add i128 [[B9]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U9]])
+; BUDGET32-NEXT: [[U10:%.*]] = add i128 [[B10]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U10]])
+; BUDGET32-NEXT: [[U11:%.*]] = add i128 [[B11]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U11]])
+; BUDGET32-NEXT: [[U12:%.*]] = add i128 [[B12]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U12]])
+; BUDGET32-NEXT: [[U13:%.*]] = add i128 [[B13]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U13]])
+; BUDGET32-NEXT: [[U14:%.*]] = add i128 [[B14]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U14]])
+; BUDGET32-NEXT: [[U15:%.*]] = add i128 [[B15]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U15]])
+; BUDGET32-NEXT: [[U16:%.*]] = add i128 [[B16]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U16]])
+; BUDGET32-NEXT: [[U17:%.*]] = add i128 [[B17]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U17]])
+; BUDGET32-NEXT: call void @bar(i128 [[B1]])
+; BUDGET32-NEXT: call void @bar(i128 [[B2]])
+; BUDGET32-NEXT: call void @bar(i128 [[B3]])
+; BUDGET32-NEXT: call void @bar(i128 [[B4]])
+; BUDGET32-NEXT: call void @bar(i128 [[B5]])
+; BUDGET32-NEXT: call void @bar(i128 [[B6]])
+; BUDGET32-NEXT: call void @bar(i128 [[B7]])
+; BUDGET32-NEXT: call void @bar(i128 [[B8]])
+; BUDGET32-NEXT: call void @bar(i128 [[B9]])
+; BUDGET32-NEXT: call void @bar(i128 [[B10]])
+; BUDGET32-NEXT: call void @bar(i128 [[B11]])
+; BUDGET32-NEXT: call void @bar(i128 [[B12]])
+; BUDGET32-NEXT: call void @bar(i128 [[B13]])
+; BUDGET32-NEXT: call void @bar(i128 [[B14]])
+; BUDGET32-NEXT: call void @bar(i128 [[B15]])
+; BUDGET32-NEXT: call void @bar(i128 [[B16]])
+; BUDGET32-NEXT: call void @bar(i128 [[B17]])
+; BUDGET32-NEXT: ret void
+;
+; ATTRS-LABEL: define void @default_occupancy(
+; ATTRS-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) {
+; ATTRS-NEXT: [[ENTRY:.*:]]
+; ATTRS-NEXT: [[S2:%.*]] = shl i128 [[S]], 1
+; ATTRS-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T1]])
+; ATTRS-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T2]])
+; ATTRS-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T3]])
+; ATTRS-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T4]])
+; ATTRS-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T5]])
+; ATTRS-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T6]])
+; ATTRS-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T7]])
+; ATTRS-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T8]])
+; ATTRS-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T9]])
+; ATTRS-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T10]])
+; ATTRS-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T11]])
+; ATTRS-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T12]])
+; ATTRS-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T13]])
+; ATTRS-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T14]])
+; ATTRS-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T15]])
+; ATTRS-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T16]])
+; ATTRS-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T17]])
+; ATTRS-NEXT: [[U1:%.*]] = add i128 [[B1]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U1]])
+; ATTRS-NEXT: [[U2:%.*]] = add i128 [[B2]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U2]])
+; ATTRS-NEXT: [[U3:%.*]] = add i128 [[B3]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U3]])
+; ATTRS-NEXT: [[U4:%.*]] = add i128 [[B4]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U4]])
+; ATTRS-NEXT: [[U5:%.*]] = add i128 [[B5]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U5]])
+; ATTRS-NEXT: [[U6:%.*]] = add i128 [[B6]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U6]])
+; ATTRS-NEXT: [[U7:%.*]] = add i128 [[B7]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U7]])
+; ATTRS-NEXT: [[U8:%.*]] = add i128 [[B8]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U8]])
+; ATTRS-NEXT: [[U9:%.*]] = add i128 [[B9]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U9]])
+; ATTRS-NEXT: [[U10:%.*]] = add i128 [[B10]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U10]])
+; ATTRS-NEXT: [[U11:%.*]] = add i128 [[B11]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U11]])
+; ATTRS-NEXT: [[U12:%.*]] = add i128 [[B12]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U12]])
+; ATTRS-NEXT: [[U13:%.*]] = add i128 [[B13]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U13]])
+; ATTRS-NEXT: [[U14:%.*]] = add i128 [[B14]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U14]])
+; ATTRS-NEXT: [[U15:%.*]] = add i128 [[B15]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U15]])
+; ATTRS-NEXT: [[U16:%.*]] = add i128 [[B16]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U16]])
+; ATTRS-NEXT: [[U17:%.*]] = add i128 [[B17]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U17]])
+; ATTRS-NEXT: call void @bar(i128 [[B1]])
+; ATTRS-NEXT: call void @bar(i128 [[B2]])
+; ATTRS-NEXT: call void @bar(i128 [[B3]])
+; ATTRS-NEXT: call void @bar(i128 [[B4]])
+; ATTRS-NEXT: call void @bar(i128 [[B5]])
+; ATTRS-NEXT: call void @bar(i128 [[B6]])
+; ATTRS-NEXT: call void @bar(i128 [[B7]])
+; ATTRS-NEXT: call void @bar(i128 [[B8]])
+; ATTRS-NEXT: call void @bar(i128 [[B9]])
+; ATTRS-NEXT: call void @bar(i128 [[B10]])
+; ATTRS-NEXT: call void @bar(i128 [[B11]])
+; ATTRS-NEXT: call void @bar(i128 [[B12]])
+; ATTRS-NEXT: call void @bar(i128 [[B13]])
+; ATTRS-NEXT: call void @bar(i128 [[B14]])
+; ATTRS-NEXT: call void @bar(i128 [[B15]])
+; ATTRS-NEXT: call void @bar(i128 [[B16]])
+; ATTRS-NEXT: call void @bar(i128 [[B17]])
+; ATTRS-NEXT: ret void
+;
+; NOFILTER-LABEL: define void @default_occupancy(
+; NOFILTER-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) {
+; NOFILTER-NEXT: [[ENTRY:.*:]]
+; NOFILTER-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T1]])
+; NOFILTER-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T2]])
+; NOFILTER-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T3]])
+; NOFILTER-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T4]])
+; NOFILTER-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T5]])
+; NOFILTER-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T6]])
+; NOFILTER-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T7]])
+; NOFILTER-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T8]])
+; NOFILTER-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T9]])
+; NOFILTER-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T10]])
+; NOFILTER-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T11]])
+; NOFILTER-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T12]])
+; NOFILTER-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T13]])
+; NOFILTER-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T14]])
+; NOFILTER-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T15]])
+; NOFILTER-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T16]])
+; NOFILTER-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T17]])
+; NOFILTER-NEXT: [[U1:%.*]] = add i128 [[T1]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U1]])
+; NOFILTER-NEXT: [[U2:%.*]] = add i128 [[T2]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U2]])
+; NOFILTER-NEXT: [[U3:%.*]] = add i128 [[T3]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U3]])
+; NOFILTER-NEXT: [[U4:%.*]] = add i128 [[T4]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U4]])
+; NOFILTER-NEXT: [[U5:%.*]] = add i128 [[T5]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U5]])
+; NOFILTER-NEXT: [[U6:%.*]] = add i128 [[T6]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U6]])
+; NOFILTER-NEXT: [[U7:%.*]] = add i128 [[T7]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U7]])
+; NOFILTER-NEXT: [[U8:%.*]] = add i128 [[T8]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U8]])
+; NOFILTER-NEXT: [[U9:%.*]] = add i128 [[T9]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U9]])
+; NOFILTER-NEXT: [[U10:%.*]] = add i128 [[T10]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U10]])
+; NOFILTER-NEXT: [[U11:%.*]] = add i128 [[T11]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U11]])
+; NOFILTER-NEXT: [[U12:%.*]] = add i128 [[T12]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U12]])
+; NOFILTER-NEXT: [[U13:%.*]] = add i128 [[T13]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U13]])
+; NOFILTER-NEXT: [[U14:%.*]] = add i128 [[T14]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U14]])
+; NOFILTER-NEXT: [[U15:%.*]] = add i128 [[T15]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U15]])
+; NOFILTER-NEXT: [[U16:%.*]] = add i128 [[T16]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U16]])
+; NOFILTER-NEXT: [[U17:%.*]] = add i128 [[T17]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U17]])
+; NOFILTER-NEXT: call void @bar(i128 [[B1]])
+; NOFILTER-NEXT: call void @bar(i128 [[B2]])
+; NOFILTER-NEXT: call void @bar(i128 [[B3]])
+; NOFILTER-NEXT: call void @bar(i128 [[B4]])
+; NOFILTER-NEXT: call void @bar(i128 [[B5]])
+; NOFILTER-NEXT: call void @bar(i128 [[B6]])
+; NOFILTER-NEXT: call void @bar(i128 [[B7]])
+; NOFILTER-NEXT: call void @bar(i128 [[B8]])
+; NOFILTER-NEXT: call void @bar(i128 [[B9]])
+; NOFILTER-NEXT: call void @bar(i128 [[B10]])
+; NOFILTER-NEXT: call void @bar(i128 [[B11]])
+; NOFILTER-NEXT: call void @bar(i128 [[B12]])
+; NOFILTER-NEXT: call void @bar(i128 [[B13]])
+; NOFILTER-NEXT: call void @bar(i128 [[B14]])
+; NOFILTER-NEXT: call void @bar(i128 [[B15]])
+; NOFILTER-NEXT: call void @bar(i128 [[B16]])
+; NOFILTER-NEXT: call void @bar(i128 [[B17]])
+; NOFILTER-NEXT: ret void
+;
+entry:
+ %s2 = shl i128 %s, 1
+ %t1 = add i128 %b1, %s
+ call void @bar(i128 %t1)
+ %t2 = add i128 %b2, %s
+ call void @bar(i128 %t2)
+ %t3 = add i128 %b3, %s
+ call void @bar(i128 %t3)
+ %t4 = add i128 %b4, %s
+ call void @bar(i128 %t4)
+ %t5 = add i128 %b5, %s
+ call void @bar(i128 %t5)
+ %t6 = add i128 %b6, %s
+ call void @bar(i128 %t6)
+ %t7 = add i128 %b7, %s
+ call void @bar(i128 %t7)
+ %t8 = add i128 %b8, %s
+ call void @bar(i128 %t8)
+ %t9 = add i128 %b9, %s
+ call void @bar(i128 %t9)
+ %t10 = add i128 %b10, %s
+ call void @bar(i128 %t10)
+ %t11 = add i128 %b11, %s
+ call void @bar(i128 %t11)
+ %t12 = add i128 %b12, %s
+ call void @bar(i128 %t12)
+ %t13 = add i128 %b13, %s
+ call void @bar(i128 %t13)
+ %t14 = add i128 %b14, %s
+ call void @bar(i128 %t14)
+ %t15 = add i128 %b15, %s
+ call void @bar(i128 %t15)
+ %t16 = add i128 %b16, %s
+ call void @bar(i128 %t16)
+ %t17 = add i128 %b17, %s
+ call void @bar(i128 %t17)
+ %u1 = add i128 %b1, %s2
+ call void @bar(i128 %u1)
+ %u2 = add i128 %b2, %s2
+ call void @bar(i128 %u2)
+ %u3 = add i128 %b3, %s2
+ call void @bar(i128 %u3)
+ %u4 = add i128 %b4, %s2
+ call void @bar(i128 %u4)
+ %u5 = add i128 %b5, %s2
+ call void @bar(i128 %u5)
+ %u6 = add i128 %b6, %s2
+ call void @bar(i128 %u6)
+ %u7 = add i128 %b7, %s2
+ call void @bar(i128 %u7)
+ %u8 = add i128 %b8, %s2
+ call void @bar(i128 %u8)
+ %u9 = add i128 %b9, %s2
+ call void @bar(i128 %u9)
+ %u10 = add i128 %b10, %s2
+ call void @bar(i128 %u10)
+ %u11 = add i128 %b11, %s2
+ call void @bar(i128 %u11)
+ %u12 = add i128 %b12, %s2
+ call void @bar(i128 %u12)
+ %u13 = add i128 %b13, %s2
+ call void @bar(i128 %u13)
+ %u14 = add i128 %b14, %s2
+ call void @bar(i128 %u14)
+ %u15 = add i128 %b15, %s2
+ call void @bar(i128 %u15)
+ %u16 = add i128 %b16, %s2
+ call void @bar(i128 %u16)
+ %u17 = add i128 %b17, %s2
+ call void @bar(i128 %u17)
+ call void @bar(i128 %b1)
+ call void @bar(i128 %b2)
+ call void @bar(i128 %b3)
+ call void @bar(i128 %b4)
+ call void @bar(i128 %b5)
+ call void @bar(i128 %b6)
+ call void @bar(i128 %b7)
+ call void @bar(i128 %b8)
+ call void @bar(i128 %b9)
+ call void @bar(i128 %b10)
+ call void @bar(i128 %b11)
+ call void @bar(i128 %b12)
+ call void @bar(i128 %b13)
+ call void @bar(i128 %b14)
+ call void @bar(i128 %b15)
+ call void @bar(i128 %b16)
+ call void @bar(i128 %b17)
+ ret void
+}
+
+; A 256-thread work group is 4 wave64s over 4 EUs, so 1 wave per EU and the full
+; 512-register budget. The same 144 now fits, so the rewrites survive.
+define void @small_work_group(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) #0 {
+; BUDGET32-LABEL: define void @small_work_group(
+; BUDGET32-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR0:[0-9]+]] {
+; BUDGET32-NEXT: [[ENTRY:.*:]]
+; BUDGET32-NEXT: [[S2:%.*]] = shl i128 [[S]], 1
+; BUDGET32-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T1]])
+; BUDGET32-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T2]])
+; BUDGET32-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T3]])
+; BUDGET32-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T4]])
+; BUDGET32-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T5]])
+; BUDGET32-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T6]])
+; BUDGET32-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T7]])
+; BUDGET32-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T8]])
+; BUDGET32-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T9]])
+; BUDGET32-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T10]])
+; BUDGET32-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T11]])
+; BUDGET32-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T12]])
+; BUDGET32-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T13]])
+; BUDGET32-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T14]])
+; BUDGET32-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T15]])
+; BUDGET32-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T16]])
+; BUDGET32-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T17]])
+; BUDGET32-NEXT: [[U1:%.*]] = add i128 [[B1]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U1]])
+; BUDGET32-NEXT: [[U2:%.*]] = add i128 [[B2]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U2]])
+; BUDGET32-NEXT: [[U3:%.*]] = add i128 [[B3]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U3]])
+; BUDGET32-NEXT: [[U4:%.*]] = add i128 [[B4]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U4]])
+; BUDGET32-NEXT: [[U5:%.*]] = add i128 [[B5]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U5]])
+; BUDGET32-NEXT: [[U6:%.*]] = add i128 [[B6]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U6]])
+; BUDGET32-NEXT: [[U7:%.*]] = add i128 [[B7]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U7]])
+; BUDGET32-NEXT: [[U8:%.*]] = add i128 [[B8]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U8]])
+; BUDGET32-NEXT: [[U9:%.*]] = add i128 [[B9]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U9]])
+; BUDGET32-NEXT: [[U10:%.*]] = add i128 [[B10]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U10]])
+; BUDGET32-NEXT: [[U11:%.*]] = add i128 [[B11]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U11]])
+; BUDGET32-NEXT: [[U12:%.*]] = add i128 [[B12]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U12]])
+; BUDGET32-NEXT: [[U13:%.*]] = add i128 [[B13]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U13]])
+; BUDGET32-NEXT: [[U14:%.*]] = add i128 [[B14]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U14]])
+; BUDGET32-NEXT: [[U15:%.*]] = add i128 [[B15]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U15]])
+; BUDGET32-NEXT: [[U16:%.*]] = add i128 [[B16]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U16]])
+; BUDGET32-NEXT: [[U17:%.*]] = add i128 [[B17]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U17]])
+; BUDGET32-NEXT: call void @bar(i128 [[B1]])
+; BUDGET32-NEXT: call void @bar(i128 [[B2]])
+; BUDGET32-NEXT: call void @bar(i128 [[B3]])
+; BUDGET32-NEXT: call void @bar(i128 [[B4]])
+; BUDGET32-NEXT: call void @bar(i128 [[B5]])
+; BUDGET32-NEXT: call void @bar(i128 [[B6]])
+; BUDGET32-NEXT: call void @bar(i128 [[B7]])
+; BUDGET32-NEXT: call void @bar(i128 [[B8]])
+; BUDGET32-NEXT: call void @bar(i128 [[B9]])
+; BUDGET32-NEXT: call void @bar(i128 [[B10]])
+; BUDGET32-NEXT: call void @bar(i128 [[B11]])
+; BUDGET32-NEXT: call void @bar(i128 [[B12]])
+; BUDGET32-NEXT: call void @bar(i128 [[B13]])
+; BUDGET32-NEXT: call void @bar(i128 [[B14]])
+; BUDGET32-NEXT: call void @bar(i128 [[B15]])
+; BUDGET32-NEXT: call void @bar(i128 [[B16]])
+; BUDGET32-NEXT: call void @bar(i128 [[B17]])
+; BUDGET32-NEXT: ret void
+;
+; ATTRS-LABEL: define void @small_work_group(
+; ATTRS-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR0:[0-9]+]] {
+; ATTRS-NEXT: [[ENTRY:.*:]]
+; ATTRS-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T1]])
+; ATTRS-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T2]])
+; ATTRS-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T3]])
+; ATTRS-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T4]])
+; ATTRS-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T5]])
+; ATTRS-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T6]])
+; ATTRS-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T7]])
+; ATTRS-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T8]])
+; ATTRS-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T9]])
+; ATTRS-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T10]])
+; ATTRS-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T11]])
+; ATTRS-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T12]])
+; ATTRS-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T13]])
+; ATTRS-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T14]])
+; ATTRS-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T15]])
+; ATTRS-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T16]])
+; ATTRS-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T17]])
+; ATTRS-NEXT: [[U1:%.*]] = add i128 [[T1]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U1]])
+; ATTRS-NEXT: [[U2:%.*]] = add i128 [[T2]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U2]])
+; ATTRS-NEXT: [[U3:%.*]] = add i128 [[T3]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U3]])
+; ATTRS-NEXT: [[U4:%.*]] = add i128 [[T4]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U4]])
+; ATTRS-NEXT: [[U5:%.*]] = add i128 [[T5]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U5]])
+; ATTRS-NEXT: [[U6:%.*]] = add i128 [[T6]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U6]])
+; ATTRS-NEXT: [[U7:%.*]] = add i128 [[T7]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U7]])
+; ATTRS-NEXT: [[U8:%.*]] = add i128 [[T8]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U8]])
+; ATTRS-NEXT: [[U9:%.*]] = add i128 [[T9]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U9]])
+; ATTRS-NEXT: [[U10:%.*]] = add i128 [[T10]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U10]])
+; ATTRS-NEXT: [[U11:%.*]] = add i128 [[T11]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U11]])
+; ATTRS-NEXT: [[U12:%.*]] = add i128 [[T12]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U12]])
+; ATTRS-NEXT: [[U13:%.*]] = add i128 [[T13]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U13]])
+; ATTRS-NEXT: [[U14:%.*]] = add i128 [[T14]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U14]])
+; ATTRS-NEXT: [[U15:%.*]] = add i128 [[T15]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U15]])
+; ATTRS-NEXT: [[U16:%.*]] = add i128 [[T16]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U16]])
+; ATTRS-NEXT: [[U17:%.*]] = add i128 [[T17]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[U17]])
+; ATTRS-NEXT: call void @bar(i128 [[B1]])
+; ATTRS-NEXT: call void @bar(i128 [[B2]])
+; ATTRS-NEXT: call void @bar(i128 [[B3]])
+; ATTRS-NEXT: call void @bar(i128 [[B4]])
+; ATTRS-NEXT: call void @bar(i128 [[B5]])
+; ATTRS-NEXT: call void @bar(i128 [[B6]])
+; ATTRS-NEXT: call void @bar(i128 [[B7]])
+; ATTRS-NEXT: call void @bar(i128 [[B8]])
+; ATTRS-NEXT: call void @bar(i128 [[B9]])
+; ATTRS-NEXT: call void @bar(i128 [[B10]])
+; ATTRS-NEXT: call void @bar(i128 [[B11]])
+; ATTRS-NEXT: call void @bar(i128 [[B12]])
+; ATTRS-NEXT: call void @bar(i128 [[B13]])
+; ATTRS-NEXT: call void @bar(i128 [[B14]])
+; ATTRS-NEXT: call void @bar(i128 [[B15]])
+; ATTRS-NEXT: call void @bar(i128 [[B16]])
+; ATTRS-NEXT: call void @bar(i128 [[B17]])
+; ATTRS-NEXT: ret void
+;
+; NOFILTER-LABEL: define void @small_work_group(
+; NOFILTER-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR0:[0-9]+]] {
+; NOFILTER-NEXT: [[ENTRY:.*:]]
+; NOFILTER-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T1]])
+; NOFILTER-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T2]])
+; NOFILTER-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T3]])
+; NOFILTER-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T4]])
+; NOFILTER-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T5]])
+; NOFILTER-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T6]])
+; NOFILTER-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T7]])
+; NOFILTER-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T8]])
+; NOFILTER-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T9]])
+; NOFILTER-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T10]])
+; NOFILTER-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T11]])
+; NOFILTER-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T12]])
+; NOFILTER-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T13]])
+; NOFILTER-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T14]])
+; NOFILTER-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T15]])
+; NOFILTER-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T16]])
+; NOFILTER-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T17]])
+; NOFILTER-NEXT: [[U1:%.*]] = add i128 [[T1]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U1]])
+; NOFILTER-NEXT: [[U2:%.*]] = add i128 [[T2]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U2]])
+; NOFILTER-NEXT: [[U3:%.*]] = add i128 [[T3]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U3]])
+; NOFILTER-NEXT: [[U4:%.*]] = add i128 [[T4]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U4]])
+; NOFILTER-NEXT: [[U5:%.*]] = add i128 [[T5]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U5]])
+; NOFILTER-NEXT: [[U6:%.*]] = add i128 [[T6]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U6]])
+; NOFILTER-NEXT: [[U7:%.*]] = add i128 [[T7]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U7]])
+; NOFILTER-NEXT: [[U8:%.*]] = add i128 [[T8]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U8]])
+; NOFILTER-NEXT: [[U9:%.*]] = add i128 [[T9]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U9]])
+; NOFILTER-NEXT: [[U10:%.*]] = add i128 [[T10]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U10]])
+; NOFILTER-NEXT: [[U11:%.*]] = add i128 [[T11]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U11]])
+; NOFILTER-NEXT: [[U12:%.*]] = add i128 [[T12]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U12]])
+; NOFILTER-NEXT: [[U13:%.*]] = add i128 [[T13]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U13]])
+; NOFILTER-NEXT: [[U14:%.*]] = add i128 [[T14]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U14]])
+; NOFILTER-NEXT: [[U15:%.*]] = add i128 [[T15]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U15]])
+; NOFILTER-NEXT: [[U16:%.*]] = add i128 [[T16]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U16]])
+; NOFILTER-NEXT: [[U17:%.*]] = add i128 [[T17]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U17]])
+; NOFILTER-NEXT: call void @bar(i128 [[B1]])
+; NOFILTER-NEXT: call void @bar(i128 [[B2]])
+; NOFILTER-NEXT: call void @bar(i128 [[B3]])
+; NOFILTER-NEXT: call void @bar(i128 [[B4]])
+; NOFILTER-NEXT: call void @bar(i128 [[B5]])
+; NOFILTER-NEXT: call void @bar(i128 [[B6]])
+; NOFILTER-NEXT: call void @bar(i128 [[B7]])
+; NOFILTER-NEXT: call void @bar(i128 [[B8]])
+; NOFILTER-NEXT: call void @bar(i128 [[B9]])
+; NOFILTER-NEXT: call void @bar(i128 [[B10]])
+; NOFILTER-NEXT: call void @bar(i128 [[B11]])
+; NOFILTER-NEXT: call void @bar(i128 [[B12]])
+; NOFILTER-NEXT: call void @bar(i128 [[B13]])
+; NOFILTER-NEXT: call void @bar(i128 [[B14]])
+; NOFILTER-NEXT: call void @bar(i128 [[B15]])
+; NOFILTER-NEXT: call void @bar(i128 [[B16]])
+; NOFILTER-NEXT: call void @bar(i128 [[B17]])
+; NOFILTER-NEXT: ret void
+;
+entry:
+ %s2 = shl i128 %s, 1
+ %t1 = add i128 %b1, %s
+ call void @bar(i128 %t1)
+ %t2 = add i128 %b2, %s
+ call void @bar(i128 %t2)
+ %t3 = add i128 %b3, %s
+ call void @bar(i128 %t3)
+ %t4 = add i128 %b4, %s
+ call void @bar(i128 %t4)
+ %t5 = add i128 %b5, %s
+ call void @bar(i128 %t5)
+ %t6 = add i128 %b6, %s
+ call void @bar(i128 %t6)
+ %t7 = add i128 %b7, %s
+ call void @bar(i128 %t7)
+ %t8 = add i128 %b8, %s
+ call void @bar(i128 %t8)
+ %t9 = add i128 %b9, %s
+ call void @bar(i128 %t9)
+ %t10 = add i128 %b10, %s
+ call void @bar(i128 %t10)
+ %t11 = add i128 %b11, %s
+ call void @bar(i128 %t11)
+ %t12 = add i128 %b12, %s
+ call void @bar(i128 %t12)
+ %t13 = add i128 %b13, %s
+ call void @bar(i128 %t13)
+ %t14 = add i128 %b14, %s
+ call void @bar(i128 %t14)
+ %t15 = add i128 %b15, %s
+ call void @bar(i128 %t15)
+ %t16 = add i128 %b16, %s
+ call void @bar(i128 %t16)
+ %t17 = add i128 %b17, %s
+ call void @bar(i128 %t17)
+ %u1 = add i128 %b1, %s2
+ call void @bar(i128 %u1)
+ %u2 = add i128 %b2, %s2
+ call void @bar(i128 %u2)
+ %u3 = add i128 %b3, %s2
+ call void @bar(i128 %u3)
+ %u4 = add i128 %b4, %s2
+ call void @bar(i128 %u4)
+ %u5 = add i128 %b5, %s2
+ call void @bar(i128 %u5)
+ %u6 = add i128 %b6, %s2
+ call void @bar(i128 %u6)
+ %u7 = add i128 %b7, %s2
+ call void @bar(i128 %u7)
+ %u8 = add i128 %b8, %s2
+ call void @bar(i128 %u8)
+ %u9 = add i128 %b9, %s2
+ call void @bar(i128 %u9)
+ %u10 = add i128 %b10, %s2
+ call void @bar(i128 %u10)
+ %u11 = add i128 %b11, %s2
+ call void @bar(i128 %u11)
+ %u12 = add i128 %b12, %s2
+ call void @bar(i128 %u12)
+ %u13 = add i128 %b13, %s2
+ call void @bar(i128 %u13)
+ %u14 = add i128 %b14, %s2
+ call void @bar(i128 %u14)
+ %u15 = add i128 %b15, %s2
+ call void @bar(i128 %u15)
+ %u16 = add i128 %b16, %s2
+ call void @bar(i128 %u16)
+ %u17 = add i128 %b17, %s2
+ call void @bar(i128 %u17)
+ call void @bar(i128 %b1)
+ call void @bar(i128 %b2)
+ call void @bar(i128 %b3)
+ call void @bar(i128 %b4)
+ call void @bar(i128 %b5)
+ call void @bar(i128 %b6)
+ call void @bar(i128 %b7)
+ call void @bar(i128 %b8)
+ call void @bar(i128 %b9)
+ call void @bar(i128 %b10)
+ call void @bar(i128 %b11)
+ call void @bar(i128 %b12)
+ call void @bar(i128 %b13)
+ call void @bar(i128 %b14)
+ call void @bar(i128 %b15)
+ call void @bar(i128 %b16)
+ call void @bar(i128 %b17)
+ ret void
+}
+
+; Same small work group, but asking for 8 waves per EU drops the budget to
+; 512/8 = 64, so 144 is over again and the rewrites are dropped. This is the
+; pair that shows waves-per-eu overriding what the work group size allows.
+define void @small_wg_high_occ(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) #1 {
+; BUDGET32-LABEL: define void @small_wg_high_occ(
+; BUDGET32-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR1:[0-9]+]] {
+; BUDGET32-NEXT: [[ENTRY:.*:]]
+; BUDGET32-NEXT: [[S2:%.*]] = shl i128 [[S]], 1
+; BUDGET32-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T1]])
+; BUDGET32-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T2]])
+; BUDGET32-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T3]])
+; BUDGET32-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T4]])
+; BUDGET32-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T5]])
+; BUDGET32-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T6]])
+; BUDGET32-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T7]])
+; BUDGET32-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T8]])
+; BUDGET32-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T9]])
+; BUDGET32-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T10]])
+; BUDGET32-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T11]])
+; BUDGET32-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T12]])
+; BUDGET32-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T13]])
+; BUDGET32-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T14]])
+; BUDGET32-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T15]])
+; BUDGET32-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T16]])
+; BUDGET32-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
+; BUDGET32-NEXT: call void @bar(i128 [[T17]])
+; BUDGET32-NEXT: [[U1:%.*]] = add i128 [[B1]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U1]])
+; BUDGET32-NEXT: [[U2:%.*]] = add i128 [[B2]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U2]])
+; BUDGET32-NEXT: [[U3:%.*]] = add i128 [[B3]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U3]])
+; BUDGET32-NEXT: [[U4:%.*]] = add i128 [[B4]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U4]])
+; BUDGET32-NEXT: [[U5:%.*]] = add i128 [[B5]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U5]])
+; BUDGET32-NEXT: [[U6:%.*]] = add i128 [[B6]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U6]])
+; BUDGET32-NEXT: [[U7:%.*]] = add i128 [[B7]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U7]])
+; BUDGET32-NEXT: [[U8:%.*]] = add i128 [[B8]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U8]])
+; BUDGET32-NEXT: [[U9:%.*]] = add i128 [[B9]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U9]])
+; BUDGET32-NEXT: [[U10:%.*]] = add i128 [[B10]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U10]])
+; BUDGET32-NEXT: [[U11:%.*]] = add i128 [[B11]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U11]])
+; BUDGET32-NEXT: [[U12:%.*]] = add i128 [[B12]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U12]])
+; BUDGET32-NEXT: [[U13:%.*]] = add i128 [[B13]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U13]])
+; BUDGET32-NEXT: [[U14:%.*]] = add i128 [[B14]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U14]])
+; BUDGET32-NEXT: [[U15:%.*]] = add i128 [[B15]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U15]])
+; BUDGET32-NEXT: [[U16:%.*]] = add i128 [[B16]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U16]])
+; BUDGET32-NEXT: [[U17:%.*]] = add i128 [[B17]], [[S2]]
+; BUDGET32-NEXT: call void @bar(i128 [[U17]])
+; BUDGET32-NEXT: call void @bar(i128 [[B1]])
+; BUDGET32-NEXT: call void @bar(i128 [[B2]])
+; BUDGET32-NEXT: call void @bar(i128 [[B3]])
+; BUDGET32-NEXT: call void @bar(i128 [[B4]])
+; BUDGET32-NEXT: call void @bar(i128 [[B5]])
+; BUDGET32-NEXT: call void @bar(i128 [[B6]])
+; BUDGET32-NEXT: call void @bar(i128 [[B7]])
+; BUDGET32-NEXT: call void @bar(i128 [[B8]])
+; BUDGET32-NEXT: call void @bar(i128 [[B9]])
+; BUDGET32-NEXT: call void @bar(i128 [[B10]])
+; BUDGET32-NEXT: call void @bar(i128 [[B11]])
+; BUDGET32-NEXT: call void @bar(i128 [[B12]])
+; BUDGET32-NEXT: call void @bar(i128 [[B13]])
+; BUDGET32-NEXT: call void @bar(i128 [[B14]])
+; BUDGET32-NEXT: call void @bar(i128 [[B15]])
+; BUDGET32-NEXT: call void @bar(i128 [[B16]])
+; BUDGET32-NEXT: call void @bar(i128 [[B17]])
+; BUDGET32-NEXT: ret void
+;
+; ATTRS-LABEL: define void @small_wg_high_occ(
+; ATTRS-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR1:[0-9]+]] {
+; ATTRS-NEXT: [[ENTRY:.*:]]
+; ATTRS-NEXT: [[S2:%.*]] = shl i128 [[S]], 1
+; ATTRS-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T1]])
+; ATTRS-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T2]])
+; ATTRS-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T3]])
+; ATTRS-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T4]])
+; ATTRS-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T5]])
+; ATTRS-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T6]])
+; ATTRS-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T7]])
+; ATTRS-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T8]])
+; ATTRS-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T9]])
+; ATTRS-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T10]])
+; ATTRS-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T11]])
+; ATTRS-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T12]])
+; ATTRS-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T13]])
+; ATTRS-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T14]])
+; ATTRS-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T15]])
+; ATTRS-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T16]])
+; ATTRS-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
+; ATTRS-NEXT: call void @bar(i128 [[T17]])
+; ATTRS-NEXT: [[U1:%.*]] = add i128 [[B1]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U1]])
+; ATTRS-NEXT: [[U2:%.*]] = add i128 [[B2]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U2]])
+; ATTRS-NEXT: [[U3:%.*]] = add i128 [[B3]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U3]])
+; ATTRS-NEXT: [[U4:%.*]] = add i128 [[B4]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U4]])
+; ATTRS-NEXT: [[U5:%.*]] = add i128 [[B5]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U5]])
+; ATTRS-NEXT: [[U6:%.*]] = add i128 [[B6]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U6]])
+; ATTRS-NEXT: [[U7:%.*]] = add i128 [[B7]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U7]])
+; ATTRS-NEXT: [[U8:%.*]] = add i128 [[B8]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U8]])
+; ATTRS-NEXT: [[U9:%.*]] = add i128 [[B9]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U9]])
+; ATTRS-NEXT: [[U10:%.*]] = add i128 [[B10]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U10]])
+; ATTRS-NEXT: [[U11:%.*]] = add i128 [[B11]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U11]])
+; ATTRS-NEXT: [[U12:%.*]] = add i128 [[B12]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U12]])
+; ATTRS-NEXT: [[U13:%.*]] = add i128 [[B13]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U13]])
+; ATTRS-NEXT: [[U14:%.*]] = add i128 [[B14]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U14]])
+; ATTRS-NEXT: [[U15:%.*]] = add i128 [[B15]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U15]])
+; ATTRS-NEXT: [[U16:%.*]] = add i128 [[B16]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U16]])
+; ATTRS-NEXT: [[U17:%.*]] = add i128 [[B17]], [[S2]]
+; ATTRS-NEXT: call void @bar(i128 [[U17]])
+; ATTRS-NEXT: call void @bar(i128 [[B1]])
+; ATTRS-NEXT: call void @bar(i128 [[B2]])
+; ATTRS-NEXT: call void @bar(i128 [[B3]])
+; ATTRS-NEXT: call void @bar(i128 [[B4]])
+; ATTRS-NEXT: call void @bar(i128 [[B5]])
+; ATTRS-NEXT: call void @bar(i128 [[B6]])
+; ATTRS-NEXT: call void @bar(i128 [[B7]])
+; ATTRS-NEXT: call void @bar(i128 [[B8]])
+; ATTRS-NEXT: call void @bar(i128 [[B9]])
+; ATTRS-NEXT: call void @bar(i128 [[B10]])
+; ATTRS-NEXT: call void @bar(i128 [[B11]])
+; ATTRS-NEXT: call void @bar(i128 [[B12]])
+; ATTRS-NEXT: call void @bar(i128 [[B13]])
+; ATTRS-NEXT: call void @bar(i128 [[B14]])
+; ATTRS-NEXT: call void @bar(i128 [[B15]])
+; ATTRS-NEXT: call void @bar(i128 [[B16]])
+; ATTRS-NEXT: call void @bar(i128 [[B17]])
+; ATTRS-NEXT: ret void
+;
+; NOFILTER-LABEL: define void @small_wg_high_occ(
+; NOFILTER-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR1:[0-9]+]] {
+; NOFILTER-NEXT: [[ENTRY:.*:]]
+; NOFILTER-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T1]])
+; NOFILTER-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T2]])
+; NOFILTER-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T3]])
+; NOFILTER-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T4]])
+; NOFILTER-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T5]])
+; NOFILTER-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T6]])
+; NOFILTER-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T7]])
+; NOFILTER-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T8]])
+; NOFILTER-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T9]])
+; NOFILTER-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T10]])
+; NOFILTER-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T11]])
+; NOFILTER-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T12]])
+; NOFILTER-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T13]])
+; NOFILTER-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T14]])
+; NOFILTER-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T15]])
+; NOFILTER-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T16]])
+; NOFILTER-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[T17]])
+; NOFILTER-NEXT: [[U1:%.*]] = add i128 [[T1]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U1]])
+; NOFILTER-NEXT: [[U2:%.*]] = add i128 [[T2]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U2]])
+; NOFILTER-NEXT: [[U3:%.*]] = add i128 [[T3]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U3]])
+; NOFILTER-NEXT: [[U4:%.*]] = add i128 [[T4]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U4]])
+; NOFILTER-NEXT: [[U5:%.*]] = add i128 [[T5]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U5]])
+; NOFILTER-NEXT: [[U6:%.*]] = add i128 [[T6]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U6]])
+; NOFILTER-NEXT: [[U7:%.*]] = add i128 [[T7]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U7]])
+; NOFILTER-NEXT: [[U8:%.*]] = add i128 [[T8]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U8]])
+; NOFILTER-NEXT: [[U9:%.*]] = add i128 [[T9]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U9]])
+; NOFILTER-NEXT: [[U10:%.*]] = add i128 [[T10]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U10]])
+; NOFILTER-NEXT: [[U11:%.*]] = add i128 [[T11]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U11]])
+; NOFILTER-NEXT: [[U12:%.*]] = add i128 [[T12]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U12]])
+; NOFILTER-NEXT: [[U13:%.*]] = add i128 [[T13]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U13]])
+; NOFILTER-NEXT: [[U14:%.*]] = add i128 [[T14]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U14]])
+; NOFILTER-NEXT: [[U15:%.*]] = add i128 [[T15]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U15]])
+; NOFILTER-NEXT: [[U16:%.*]] = add i128 [[T16]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U16]])
+; NOFILTER-NEXT: [[U17:%.*]] = add i128 [[T17]], [[S]]
+; NOFILTER-NEXT: call void @bar(i128 [[U17]])
+; NOFILTER-NEXT: call void @bar(i128 [[B1]])
+; NOFILTER-NEXT: call void @bar(i128 [[B2]])
+; NOFILTER-NEXT: call void @bar(i128 [[B3]])
+; NOFILTER-NEXT: call void @bar(i128 [[B4]])
+; NOFILTER-NEXT: call void @bar(i128 [[B5]])
+; NOFILTER-NEXT: call void @bar(i128 [[B6]])
+; NOFILTER-NEXT: call void @bar(i128 [[B7]])
+; NOFILTER-NEXT: call void @bar(i128 [[B8]])
+; NOFILTER-NEXT: call void @bar(i128 [[B9]])
+; NOFILTER-NEXT: call void @bar(i128 [[B10]])
+; NOFILTER-NEXT: call void @bar(i128 [[B11]])
+; NOFILTER-NEXT: call void @bar(i128 [[B12]])
+; NOFILTER-NEXT: call void @bar(i128 [[B13]])
+; NOFILTER-NEXT: call void @bar(i128 [[B14]])
+; NOFILTER-NEXT: call void @bar(i128 [[B15]])
+; NOFILTER-NEXT: call void @bar(i128 [[B16]])
+; NOFILTER-NEXT: call void @bar(i128 [[B17]])
+; NOFILTER-NEXT: ret void
+;
+entry:
+ %s2 = shl i128 %s, 1
+ %t1 = add i128 %b1, %s
+ call void @bar(i128 %t1)
+ %t2 = add i128 %b2, %s
+ call void @bar(i128 %t2)
+ %t3 = add i128 %b3, %s
+ call void @bar(i128 %t3)
+ %t4 = add i128 %b4, %s
+ call void @bar(i128 %t4)
+ %t5 = add i128 %b5, %s
+ call void @bar(i128 %t5)
+ %t6 = add i128 %b6, %s
+ call void @bar(i128 %t6)
+ %t7 = add i128 %b7, %s
+ call void @bar(i128 %t7)
+ %t8 = add i128 %b8, %s
+ call void @bar(i128 %t8)
+ %t9 = add i128 %b9, %s
+ call void @bar(i128 %t9)
+ %t10 = add i128 %b10, %s
+ call void @bar(i128 %t10)
+ %t11 = add i128 %b11, %s
+ call void @bar(i128 %t11)
+ %t12 = add i128 %b12, %s
+ call void @bar(i128 %t12)
+ %t13 = add i128 %b13, %s
+ call void @bar(i128 %t13)
+ %t14 = add i128 %b14, %s
+ call void @bar(i128 %t14)
+ %t15 = add i128 %b15, %s
+ call void @bar(i128 %t15)
+ %t16 = add i128 %b16, %s
+ call void @bar(i128 %t16)
+ %t17 = add i128 %b17, %s
+ call void @bar(i128 %t17)
+ %u1 = add i128 %b1, %s2
+ call void @bar(i128 %u1)
+ %u2 = add i128 %b2, %s2
+ call void @bar(i128 %u2)
+ %u3 = add i128 %b3, %s2
+ call void @bar(i128 %u3)
+ %u4 = add i128 %b4, %s2
+ call void @bar(i128 %u4)
+ %u5 = add i128 %b5, %s2
+ call void @bar(i128 %u5)
+ %u6 = add i128 %b6, %s2
+ call void @bar(i128 %u6)
+ %u7 = add i128 %b7, %s2
+ call void @bar(i128 %u7)
+ %u8 = add i128 %b8, %s2
+ call void @bar(i128 %u8)
+ %u9 = add i128 %b9, %s2
+ call void @bar(i128 %u9)
+ %u10 = add i128 %b10, %s2
+ call void @bar(i128 %u10)
+ %u11 = add i128 %b11, %s2
+ call void @bar(i128 %u11)
+ %u12 = add i128 %b12, %s2
+ call void @bar(i128 %u12)
+ %u13 = add i128 %b13, %s2
+ call void @bar(i128 %u13)
+ %u14 = add i128 %b14, %s2
+ call void @bar(i128 %u14)
+ %u15 = add i128 %b15, %s2
+ call void @bar(i128 %u15)
+ %u16 = add i128 %b16, %s2
+ call void @bar(i128 %u16)
+ %u17 = add i128 %b17, %s2
+ call void @bar(i128 %u17)
+ call void @bar(i128 %b1)
+ call void @bar(i128 %b2)
+ call void @bar(i128 %b3)
+ call void @bar(i128 %b4)
+ call void @bar(i128 %b5)
+ call void @bar(i128 %b6)
+ call void @bar(i128 %b7)
+ call void @bar(i128 %b8)
+ call void @bar(i128 %b9)
+ call void @bar(i128 %b10)
+ call void @bar(i128 %b11)
+ call void @bar(i128 %b12)
+ call void @bar(i128 %b13)
+ call void @bar(i128 %b14)
+ call void @bar(i128 %b15)
+ call void @bar(i128 %b16)
+ call void @bar(i128 %b17)
+ ret void
+}
+
+attributes #0 = { "amdgpu-flat-work-group-size"="1,256" }
+attributes #1 = { "amdgpu-flat-work-group-size"="1,256" "amdgpu-waves-per-eu"="8" }
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
deleted file mode 100644
index 42aa95b8dd006..0000000000000
--- a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
+++ /dev/null
@@ -1,94 +0,0 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt < %s -passes=slsr -S -slsr-basis-distance-threshold=96 | FileCheck %s --check-prefixes=CHECK,REWRITE
-; RUN: opt < %s -passes=slsr -S -slsr-basis-distance-threshold=2 | FileCheck %s --check-prefixes=CHECK,SKIP
-
-; The register-pressure cost model skips a rewrite only when BOTH:
-; 1. an operand of the candidate is used by a non-rewritable user in
-; another block, and
-; 2. the basis' last same-block use is farther than
-; -slsr-basis-distance-threshold from the candidate.
-;
-; Rewriting of %t2 = add i32 %t1, %s based on basis %t1 shouldn't happen when slsr-basis-distance-threshold is 2.
-; Distance from %t1's last use to %t2 is 5 > 2 and %s2 continues to be used in "next".
-
-target datalayout = "e-i64:64-v16:16-v32:32-n16:32:64"
-
-declare void @foo(i32)
-declare void @use(i32)
-
-define void @basis_too_far(i32 %b, i32 %s) {
-; REWRITE-LABEL: define void @basis_too_far(
-; REWRITE-SAME: i32 [[B:%.*]], i32 [[S:%.*]]) {
-; REWRITE-NEXT: [[ENTRY:.*:]]
-; REWRITE-NEXT: [[T1:%.*]] = add i32 [[B]], [[S]]
-; REWRITE-NEXT: call void @foo(i32 [[T1]])
-; REWRITE-NEXT: call void @foo(i32 [[B]])
-; REWRITE-NEXT: call void @foo(i32 [[B]])
-; REWRITE-NEXT: call void @foo(i32 [[B]])
-; REWRITE-NEXT: [[S2:%.*]] = shl i32 [[S]], 1
-; REWRITE-NEXT: [[T2:%.*]] = add i32 [[T1]], [[S]]
-; REWRITE-NEXT: call void @foo(i32 [[T2]])
-; REWRITE-NEXT: br label %[[NEXT:.*]]
-; REWRITE: [[NEXT]]:
-; REWRITE-NEXT: call void @use(i32 [[S2]])
-; REWRITE-NEXT: ret void
-;
-; SKIP-LABEL: define void @basis_too_far(
-; SKIP-SAME: i32 [[B:%.*]], i32 [[S:%.*]]) {
-; SKIP-NEXT: [[ENTRY:.*:]]
-; SKIP-NEXT: [[T1:%.*]] = add i32 [[B]], [[S]]
-; SKIP-NEXT: call void @foo(i32 [[T1]])
-; SKIP-NEXT: call void @foo(i32 [[B]])
-; SKIP-NEXT: call void @foo(i32 [[B]])
-; SKIP-NEXT: call void @foo(i32 [[B]])
-; SKIP-NEXT: [[S2:%.*]] = shl i32 [[S]], 1
-; SKIP-NEXT: [[T2:%.*]] = add i32 [[B]], [[S2]]
-; SKIP-NEXT: call void @foo(i32 [[T2]])
-; SKIP-NEXT: br label %[[NEXT:.*]]
-; SKIP: [[NEXT]]:
-; SKIP-NEXT: call void @use(i32 [[S2]])
-; SKIP-NEXT: ret void
-;
-entry:
- %t1 = add i32 %b, %s
- call void @foo(i32 %t1)
- call void @foo(i32 %b)
- call void @foo(i32 %b)
- call void @foo(i32 %b)
- %s2 = shl i32 %s, 1
- %t2 = add i32 %b, %s2
- call void @foo(i32 %t2)
- br label %next
-
-next:
- call void @use(i32 %s2)
- ret void
-}
-
-define void @same_block_operand(i32 %b, i32 %s) {
-; CHECK-LABEL: define void @same_block_operand(
-; CHECK-SAME: i32 [[B:%.*]], i32 [[S:%.*]]) {
-; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[T1:%.*]] = add i32 [[B]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T1]])
-; CHECK-NEXT: call void @foo(i32 [[B]])
-; CHECK-NEXT: call void @foo(i32 [[B]])
-; CHECK-NEXT: call void @foo(i32 [[B]])
-; CHECK-NEXT: [[S2:%.*]] = shl i32 [[S]], 1
-; CHECK-NEXT: [[T2:%.*]] = add i32 [[T1]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T2]])
-; CHECK-NEXT: call void @use(i32 [[S2]])
-; CHECK-NEXT: ret void
-;
-entry:
- %t1 = add i32 %b, %s
- call void @foo(i32 %t1)
- call void @foo(i32 %b)
- call void @foo(i32 %b)
- call void @foo(i32 %b)
- %s2 = shl i32 %s, 1
- %t2 = add i32 %b, %s2
- call void @foo(i32 %t2)
- call void @use(i32 %s2)
- ret void
-}
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
new file mode 100644
index 0000000000000..c534ecb8a3fda
--- /dev/null
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
@@ -0,0 +1,296 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=slsr -S | FileCheck %s --check-prefixes=CHECK,NOBUDGET
+; RUN: opt < %s -passes=slsr -S -slsr-reg-budget=32 | FileCheck %s --check-prefixes=CHECK,BUDGET
+
+; The register-pressure filter needs a register budget to compare against, and
+; the generic TargetTransformInfo has none: getRegisterBudget() returns
+; std::nullopt unless a target implements it. There is no target triple here, so
+; the filter returns early without even running its liveness or pressure
+; analyses, and every rewrite stands no matter how much pressure it adds.
+;
+; @many_bases_overlapping is the same shape used in AMDGPU/slsr-rp-filter.ll:
+; 17 distinct bases whose rewrites take the block's peak pressure from 20 to 36
+; registers. Passing -slsr-reg-budget=32 supplies the missing budget, and with
+; the 0.9 safe fraction leaving 28 registers the rewrites are then dropped.
+;
+; So the two runs differ only in whether a budget exists, which is what pins the
+; early return down: without one the filter is inert on targets that cannot
+; report a register count.
+
+target datalayout = "e-i64:64-v16:16-v32:32-n16:32:64"
+
+declare void @foo(i32)
+
+define void @many_bases_overlapping(i32 %s, i32 %b1, i32 %b2, i32 %b3, i32 %b4, i32 %b5, i32 %b6, i32 %b7, i32 %b8, i32 %b9, i32 %b10, i32 %b11, i32 %b12, i32 %b13, i32 %b14, i32 %b15, i32 %b16, i32 %b17) {
+; NOBUDGET-LABEL: define void @many_bases_overlapping(
+; NOBUDGET-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; NOBUDGET-NEXT: [[ENTRY:.*:]]
+; NOBUDGET-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T1]])
+; NOBUDGET-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T2]])
+; NOBUDGET-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T3]])
+; NOBUDGET-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T4]])
+; NOBUDGET-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T5]])
+; NOBUDGET-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T6]])
+; NOBUDGET-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T7]])
+; NOBUDGET-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T8]])
+; NOBUDGET-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T9]])
+; NOBUDGET-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T10]])
+; NOBUDGET-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T11]])
+; NOBUDGET-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T12]])
+; NOBUDGET-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T13]])
+; NOBUDGET-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T14]])
+; NOBUDGET-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T15]])
+; NOBUDGET-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T16]])
+; NOBUDGET-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[T17]])
+; NOBUDGET-NEXT: [[U1:%.*]] = add i32 [[T1]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U1]])
+; NOBUDGET-NEXT: [[U2:%.*]] = add i32 [[T2]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U2]])
+; NOBUDGET-NEXT: [[U3:%.*]] = add i32 [[T3]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U3]])
+; NOBUDGET-NEXT: [[U4:%.*]] = add i32 [[T4]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U4]])
+; NOBUDGET-NEXT: [[U5:%.*]] = add i32 [[T5]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U5]])
+; NOBUDGET-NEXT: [[U6:%.*]] = add i32 [[T6]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U6]])
+; NOBUDGET-NEXT: [[U7:%.*]] = add i32 [[T7]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U7]])
+; NOBUDGET-NEXT: [[U8:%.*]] = add i32 [[T8]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U8]])
+; NOBUDGET-NEXT: [[U9:%.*]] = add i32 [[T9]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U9]])
+; NOBUDGET-NEXT: [[U10:%.*]] = add i32 [[T10]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U10]])
+; NOBUDGET-NEXT: [[U11:%.*]] = add i32 [[T11]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U11]])
+; NOBUDGET-NEXT: [[U12:%.*]] = add i32 [[T12]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U12]])
+; NOBUDGET-NEXT: [[U13:%.*]] = add i32 [[T13]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U13]])
+; NOBUDGET-NEXT: [[U14:%.*]] = add i32 [[T14]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U14]])
+; NOBUDGET-NEXT: [[U15:%.*]] = add i32 [[T15]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U15]])
+; NOBUDGET-NEXT: [[U16:%.*]] = add i32 [[T16]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U16]])
+; NOBUDGET-NEXT: [[U17:%.*]] = add i32 [[T17]], [[S]]
+; NOBUDGET-NEXT: call void @foo(i32 [[U17]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B1]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B2]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B3]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B4]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B5]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B6]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B7]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B8]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B9]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B10]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B11]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B12]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B13]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B14]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B15]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B16]])
+; NOBUDGET-NEXT: call void @foo(i32 [[B17]])
+; NOBUDGET-NEXT: ret void
+;
+; BUDGET-LABEL: define void @many_bases_overlapping(
+; BUDGET-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; BUDGET-NEXT: [[ENTRY:.*:]]
+; BUDGET-NEXT: [[S2:%.*]] = shl i32 [[S]], 1
+; BUDGET-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T1]])
+; BUDGET-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T2]])
+; BUDGET-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T3]])
+; BUDGET-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T4]])
+; BUDGET-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T5]])
+; BUDGET-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T6]])
+; BUDGET-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T7]])
+; BUDGET-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T8]])
+; BUDGET-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T9]])
+; BUDGET-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T10]])
+; BUDGET-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T11]])
+; BUDGET-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T12]])
+; BUDGET-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T13]])
+; BUDGET-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T14]])
+; BUDGET-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T15]])
+; BUDGET-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T16]])
+; BUDGET-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
+; BUDGET-NEXT: call void @foo(i32 [[T17]])
+; BUDGET-NEXT: [[U1:%.*]] = add i32 [[B1]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U1]])
+; BUDGET-NEXT: [[U2:%.*]] = add i32 [[B2]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U2]])
+; BUDGET-NEXT: [[U3:%.*]] = add i32 [[B3]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U3]])
+; BUDGET-NEXT: [[U4:%.*]] = add i32 [[B4]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U4]])
+; BUDGET-NEXT: [[U5:%.*]] = add i32 [[B5]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U5]])
+; BUDGET-NEXT: [[U6:%.*]] = add i32 [[B6]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U6]])
+; BUDGET-NEXT: [[U7:%.*]] = add i32 [[B7]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U7]])
+; BUDGET-NEXT: [[U8:%.*]] = add i32 [[B8]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U8]])
+; BUDGET-NEXT: [[U9:%.*]] = add i32 [[B9]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U9]])
+; BUDGET-NEXT: [[U10:%.*]] = add i32 [[B10]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U10]])
+; BUDGET-NEXT: [[U11:%.*]] = add i32 [[B11]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U11]])
+; BUDGET-NEXT: [[U12:%.*]] = add i32 [[B12]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U12]])
+; BUDGET-NEXT: [[U13:%.*]] = add i32 [[B13]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U13]])
+; BUDGET-NEXT: [[U14:%.*]] = add i32 [[B14]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U14]])
+; BUDGET-NEXT: [[U15:%.*]] = add i32 [[B15]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U15]])
+; BUDGET-NEXT: [[U16:%.*]] = add i32 [[B16]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U16]])
+; BUDGET-NEXT: [[U17:%.*]] = add i32 [[B17]], [[S2]]
+; BUDGET-NEXT: call void @foo(i32 [[U17]])
+; BUDGET-NEXT: call void @foo(i32 [[B1]])
+; BUDGET-NEXT: call void @foo(i32 [[B2]])
+; BUDGET-NEXT: call void @foo(i32 [[B3]])
+; BUDGET-NEXT: call void @foo(i32 [[B4]])
+; BUDGET-NEXT: call void @foo(i32 [[B5]])
+; BUDGET-NEXT: call void @foo(i32 [[B6]])
+; BUDGET-NEXT: call void @foo(i32 [[B7]])
+; BUDGET-NEXT: call void @foo(i32 [[B8]])
+; BUDGET-NEXT: call void @foo(i32 [[B9]])
+; BUDGET-NEXT: call void @foo(i32 [[B10]])
+; BUDGET-NEXT: call void @foo(i32 [[B11]])
+; BUDGET-NEXT: call void @foo(i32 [[B12]])
+; BUDGET-NEXT: call void @foo(i32 [[B13]])
+; BUDGET-NEXT: call void @foo(i32 [[B14]])
+; BUDGET-NEXT: call void @foo(i32 [[B15]])
+; BUDGET-NEXT: call void @foo(i32 [[B16]])
+; BUDGET-NEXT: call void @foo(i32 [[B17]])
+; BUDGET-NEXT: ret void
+;
+entry:
+ %s2 = shl i32 %s, 1
+ %t1 = add i32 %b1, %s
+ call void @foo(i32 %t1)
+ %t2 = add i32 %b2, %s
+ call void @foo(i32 %t2)
+ %t3 = add i32 %b3, %s
+ call void @foo(i32 %t3)
+ %t4 = add i32 %b4, %s
+ call void @foo(i32 %t4)
+ %t5 = add i32 %b5, %s
+ call void @foo(i32 %t5)
+ %t6 = add i32 %b6, %s
+ call void @foo(i32 %t6)
+ %t7 = add i32 %b7, %s
+ call void @foo(i32 %t7)
+ %t8 = add i32 %b8, %s
+ call void @foo(i32 %t8)
+ %t9 = add i32 %b9, %s
+ call void @foo(i32 %t9)
+ %t10 = add i32 %b10, %s
+ call void @foo(i32 %t10)
+ %t11 = add i32 %b11, %s
+ call void @foo(i32 %t11)
+ %t12 = add i32 %b12, %s
+ call void @foo(i32 %t12)
+ %t13 = add i32 %b13, %s
+ call void @foo(i32 %t13)
+ %t14 = add i32 %b14, %s
+ call void @foo(i32 %t14)
+ %t15 = add i32 %b15, %s
+ call void @foo(i32 %t15)
+ %t16 = add i32 %b16, %s
+ call void @foo(i32 %t16)
+ %t17 = add i32 %b17, %s
+ call void @foo(i32 %t17)
+ %u1 = add i32 %b1, %s2
+ call void @foo(i32 %u1)
+ %u2 = add i32 %b2, %s2
+ call void @foo(i32 %u2)
+ %u3 = add i32 %b3, %s2
+ call void @foo(i32 %u3)
+ %u4 = add i32 %b4, %s2
+ call void @foo(i32 %u4)
+ %u5 = add i32 %b5, %s2
+ call void @foo(i32 %u5)
+ %u6 = add i32 %b6, %s2
+ call void @foo(i32 %u6)
+ %u7 = add i32 %b7, %s2
+ call void @foo(i32 %u7)
+ %u8 = add i32 %b8, %s2
+ call void @foo(i32 %u8)
+ %u9 = add i32 %b9, %s2
+ call void @foo(i32 %u9)
+ %u10 = add i32 %b10, %s2
+ call void @foo(i32 %u10)
+ %u11 = add i32 %b11, %s2
+ call void @foo(i32 %u11)
+ %u12 = add i32 %b12, %s2
+ call void @foo(i32 %u12)
+ %u13 = add i32 %b13, %s2
+ call void @foo(i32 %u13)
+ %u14 = add i32 %b14, %s2
+ call void @foo(i32 %u14)
+ %u15 = add i32 %b15, %s2
+ call void @foo(i32 %u15)
+ %u16 = add i32 %b16, %s2
+ call void @foo(i32 %u16)
+ %u17 = add i32 %b17, %s2
+ call void @foo(i32 %u17)
+ call void @foo(i32 %b1)
+ call void @foo(i32 %b2)
+ call void @foo(i32 %b3)
+ call void @foo(i32 %b4)
+ call void @foo(i32 %b5)
+ call void @foo(i32 %b6)
+ call void @foo(i32 %b7)
+ call void @foo(i32 %b8)
+ call void @foo(i32 %b9)
+ call void @foo(i32 %b10)
+ call void @foo(i32 %b11)
+ call void @foo(i32 %b12)
+ call void @foo(i32 %b13)
+ call void @foo(i32 %b14)
+ call void @foo(i32 %b15)
+ call void @foo(i32 %b16)
+ call void @foo(i32 %b17)
+ ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK: {{.*}}
>From 888f93e45cd34e5ccc15390d3febccf6d9a4795b Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Wed, 2 Sep 2026 23:34:06 +0000
Subject: [PATCH 08/13] Remove trivial knobs, shorten tests
---
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 9 +-
.../Scalar/StraightLineStrengthReduce.cpp | 51 +-
.../AMDGPU/slsr-rp-filter.ll | 1888 ++---------------
.../slsr-rp-filter.ll | 384 +---
4 files changed, 298 insertions(+), 2034 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 35a0d63e77f54..16555675edda3 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -311,13 +311,8 @@ unsigned GCNTTIImpl::getNumberOfRegisters(unsigned RCID) const {
std::optional<unsigned> GCNTTIImpl::getRegisterBudget(const Function &F) const {
// Report the VGPR budget implied by the occupancy F is compiled for. Callers
- // comparing a single lumped pressure number against this are conservative on
- // the SGPR side, which is intentional: whether a value lands in an SGPR or a
- // VGPR depends on divergence, not on its type.
- //
- // On GFX90A this is the combined VGPR+AGPR budget; see getMaxNumVectorRegs
- // for the split. In dynamic VGPR mode "amdgpu-waves-per-eu" implies no VGPR
- // limit at all, so this degrades to the full register file.
+ // comparing a single lumped pressure number against this should be
+ // conservative on the SGPR side, which is intentional.
return ST->getMaxNumVGPRs(F);
}
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index ae39823827087..45dc1942e1bb7 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -125,32 +125,11 @@ static cl::opt<bool>
EnablePoisonReuseGuard("enable-poison-reuse-guard", cl::init(true),
cl::desc("Enable poison-reuse guard"));
-// RPFilter targets one pathological shape: a block holding many distinct
-// bases. Each basis contributes one extended live range no matter how many
-// candidates are rewritten against it, so it is the number of distinct bases,
-// not the number of candidates, that tracks how many new concurrent live
-// ranges SLSR would create. Below this count no block in the function can
-// exhibit the pathology, and the liveness and pressure analyses are skipped.
-static constexpr unsigned MinDistinctBasesToFilter = 16;
-
static cl::opt<bool> EnableRPFilter(
"slsr-rp-filter", cl::init(true), cl::Hidden,
cl::desc("SLSR: skip rewrites in blocks where they would push register "
"pressure past the target's register budget"));
-static cl::opt<unsigned> SLSRRegBudget(
- "slsr-reg-budget", cl::init(0), cl::Hidden,
- cl::desc("SLSR: override the register budget reported by TTI"));
-
-static cl::opt<double> SLSRRPSafeFraction(
- "slsr-rp-safe-fraction", cl::init(0.9), cl::Hidden,
- cl::desc("SLSR: fraction of the register budget treated as safe"));
-
-static cl::opt<unsigned> SLSRRPAbsDelta(
- "slsr-rp-abs-delta", cl::init(4), cl::Hidden,
- cl::desc("SLSR: pressure increase tolerated in a block that is already "
- "over the register budget"));
-
STATISTIC(NumSCEVCandidateBasisDifferences,
"Number of candidate-basis SCEV differences computed by SLSR");
STATISTIC(NumRPFilteredBlocks,
@@ -1447,10 +1426,15 @@ bool StraightLineStrengthReduceLegacyPass::runOnFunction(Function &F) {
namespace {
-// TODO: Currently, I am considering (Basis, Cand) pair that are both in the
-// same BB.
-// The restriction may not be needed.
class RPFilter {
+ // RPFilter targets one pathological shape: a block holding many distinct
+ // bases. Each basis contributes one extended live range no matter how many
+ // candidates are rewritten against it, so it is the number of distinct bases,
+ // not the number of candidates, that tracks how many new concurrent live
+ // ranges SLSR would create. Below this count no block in the function can
+ // exhibit the pathology, and the liveness and pressure analyses are skipped.
+ static constexpr unsigned MinDistinctBasesToFilter = 16;
+
public:
using Candidate = StraightLineStrengthReduce::Candidate;
@@ -1535,8 +1519,6 @@ class RPFilter {
}
std::optional<unsigned> getRegisterBudget() const {
- if (SLSRRegBudget > 0)
- return SLSRRegBudget.getValue();
std::optional<unsigned> Budget = TTI->getRegisterBudget(*F);
if (Budget && *Budget == 0)
return std::nullopt;
@@ -1550,6 +1532,7 @@ class RPFilter {
unsigned Budget) const {
// Leave the allocator some slack: it also has to satisfy register class
// and ABI constraints that this estimate knows nothing about.
+ constexpr double SLSRRPSafeFraction = 0.9;
unsigned SafeBudget = static_cast<unsigned>(Budget * SLSRRPSafeFraction);
// There is headroom, so however much the rewrite adds is irrelevant.
@@ -1562,6 +1545,7 @@ class RPFilter {
// Already over budget. SLSR can still lower pressure here, so only refuse
// rewrites that make it meaningfully worse.
+ constexpr unsigned SLSRRPAbsDelta = 4;
return After > Before && After - Before > SLSRRPAbsDelta;
}
@@ -1701,10 +1685,6 @@ class RPFilter {
});
}
- bool isDebugBlock(const BasicBlock &BB) const {
- return BB.getName() == "for.cond.cleanup";
- }
-
// Return true if I is a candidate and its basis is in the same bb, false
// otherwise.
bool insertBasisIfCand(const Instruction *I, ValueSet &LiveSetWithSLSR,
@@ -1714,6 +1694,11 @@ class RPFilter {
return false;
// I is a Candidate
+ // We consider here only the case both Cand and Basis are in the same BB.
+ // Therefore, if an original operand of Cand is a liveout, it means it was
+ // used outside the BB, and will stay so. Likewise, if a Basis is a liveout,
+ // it means it was used outside the BB, and will stay so. SLSR does not
+ // delete existing defs or create new defs.
const Instruction *Basis = It->second->Basis->Ins;
if (Basis->getParent() == I->getParent()) {
LiveSetWithSLSR.insert(Basis);
@@ -1746,12 +1731,6 @@ class RPFilter {
// to
// Basis = ..
// Cand = f(Basis, Delta, // possibly some of the original operands ..)
- //
- // We consider here only the case both Cand and Basis are in the same BB.
- // Therefore, if an original operand of Cand is a liveout, it means it was
- // used outside the BB, and will stay so. Likewise, if a Basis is a liveout,
- // it means it was used outside the BB, and will stay so. SLSR does not
- // delete existing defs or create new defs.
ValueSet LiveSetWithSLSR = LiveOut;
unsigned MaxWWithSLSR = MaxW;
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
index 24ae488d9a7f3..a5f86a1a42ce4 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
@@ -1,7 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt < %s -mtriple=amdgpu9.50-amd-amdhsa -passes=slsr -S -slsr-reg-budget=32 | FileCheck %s --check-prefixes=CHECK,BUDGET32
-; RUN: opt < %s -mtriple=amdgpu9.50-amd-amdhsa -passes=slsr -S | FileCheck %s --check-prefixes=CHECK,ATTRS
-; RUN: opt < %s -mtriple=amdgpu9.50-amd-amdhsa -passes=slsr -S -slsr-rp-filter=false | FileCheck %s --check-prefixes=CHECK,NOFILTER
+; RUN: opt < %s -mtriple=amdgpu-amdhsa-amd -passes=slsr -S | FileCheck %s
; SLSR rewrites a candidate as its basis plus a delta, which keeps the basis live
; down to the candidate. The register-pressure filter drops a block's rewrites
@@ -9,769 +7,32 @@
; allocate.
;
; The budget comes from TTI, which on AMDGPU reports the VGPR count implied by
-; the occupancy the function is compiled for, and -slsr-reg-budget overrides it.
-; -slsr-rp-safe-fraction (0.9 by default) is applied on top, so a budget of 32
-; leaves 28 usable registers. The filter does nothing unless -slsr-rp-filter is
-; passed, and it only examines functions where some block holds more than 16
-; distinct bases.
+; the occupancy the function is compiled for. This triple selects the generic
+; GPU, which has 256 VGPRs in total. Waves per EU follows from the flat work
+; group size, the budget is those 256 registers divided by the waves that must
+; share an EU, and the filter treats 0.9 of the budget as usable.
;
-; @many_bases_overlapping has 17 bases and goes from 20 to 36 registers, crossing
-; the 28-register safe budget, so its rewrites are dropped. @many_bases_adjacent
-; has the same 17 bases, but each %u<k> immediately follows its %t<k>, so the
-; extended ranges never coexist and the peak only reaches 21, which fits.
+; Both functions below have byte-identical bodies: 17 distinct bases whose
+; rewrites take the block's peak pressure from 80 registers to 144. The only
+; difference is the occupancy attribute, so the budget is the only variable and
+; the comparison isolates it.
;
-; @four_bases holds only 4 bases, below the gate, so the filter never looks at it
-; and its rewrites stand. Its i128 values put the block at 28 registers rising to
-; 44, which would cross the 28-register safe budget if it were examined, so the
-; gate really is the only thing sparing it.
+; @peak_above_budget no attribute. The work group size defaults to 1024
+; threads, which is 16 wave64s over 4 EUs, so 4 waves per
+; EU and a budget of 256/4 = 64 registers, 57 usable.
+; 144 is above 57, so the rewrites are dropped.
;
-; The i32 functions carry no occupancy attributes, so their flat work group size
-; defaults to 1024, which is 16 wave64s spread over 4 EUs, hence 4 waves per EU
-; and a budget of 512/4 = 128 registers. Their peak of 36 fits, so the ATTRS run
-; leaves them alone and only the forced budget of 32 filters them.
-;
-; The i128 functions at the end raise the peak to 144 so the real budget decides
-; without any override. All three have identical bodies and identical pressure,
-; so the attributes are the only variable:
-;
-; @default_occupancy no attributes 4 waves/EU budget 128 filtered
-; @small_work_group flat-work-group-size 1 wave/EU budget 512 kept
-; @small_wg_high_occ + waves-per-eu=8 8 waves/EU budget 64 filtered
-;
-; The first pair shows flat-work-group-size raising the budget, and the second
-; shows waves-per-eu lowering it again. NOFILTER switches the analysis off.
+; @peak_below_budget 256 threads is 4 wave64s over 4 EUs, so 1 wave per EU
+; and a budget of the full 256 registers, 230 usable.
+; The same 144 is below 230, so the rewrites survive.
-declare void @foo(i32)
declare void @bar(i128)
-define void @many_bases_overlapping(i32 %s, i32 %b1, i32 %b2, i32 %b3, i32 %b4, i32 %b5, i32 %b6, i32 %b7, i32 %b8, i32 %b9, i32 %b10, i32 %b11, i32 %b12, i32 %b13, i32 %b14, i32 %b15, i32 %b16, i32 %b17) {
-; FILTER-LABEL: define void @many_bases_overlapping(
-; FILTER-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; FILTER-NEXT: [[ENTRY:.*:]]
-; FILTER-NEXT: [[S2:%.*]] = shl i32 [[S]], 1
-; FILTER-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T1]])
-; FILTER-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T2]])
-; FILTER-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T3]])
-; FILTER-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T4]])
-; FILTER-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T5]])
-; FILTER-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T6]])
-; FILTER-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T7]])
-; FILTER-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T8]])
-; FILTER-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T9]])
-; FILTER-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T10]])
-; FILTER-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T11]])
-; FILTER-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T12]])
-; FILTER-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T13]])
-; FILTER-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T14]])
-; FILTER-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T15]])
-; FILTER-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T16]])
-; FILTER-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
-; FILTER-NEXT: call void @foo(i32 [[T17]])
-; FILTER-NEXT: [[U1:%.*]] = add i32 [[B1]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U1]])
-; FILTER-NEXT: [[U2:%.*]] = add i32 [[B2]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U2]])
-; FILTER-NEXT: [[U3:%.*]] = add i32 [[B3]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U3]])
-; FILTER-NEXT: [[U4:%.*]] = add i32 [[B4]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U4]])
-; FILTER-NEXT: [[U5:%.*]] = add i32 [[B5]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U5]])
-; FILTER-NEXT: [[U6:%.*]] = add i32 [[B6]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U6]])
-; FILTER-NEXT: [[U7:%.*]] = add i32 [[B7]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U7]])
-; FILTER-NEXT: [[U8:%.*]] = add i32 [[B8]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U8]])
-; FILTER-NEXT: [[U9:%.*]] = add i32 [[B9]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U9]])
-; FILTER-NEXT: [[U10:%.*]] = add i32 [[B10]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U10]])
-; FILTER-NEXT: [[U11:%.*]] = add i32 [[B11]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U11]])
-; FILTER-NEXT: [[U12:%.*]] = add i32 [[B12]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U12]])
-; FILTER-NEXT: [[U13:%.*]] = add i32 [[B13]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U13]])
-; FILTER-NEXT: [[U14:%.*]] = add i32 [[B14]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U14]])
-; FILTER-NEXT: [[U15:%.*]] = add i32 [[B15]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U15]])
-; FILTER-NEXT: [[U16:%.*]] = add i32 [[B16]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U16]])
-; FILTER-NEXT: [[U17:%.*]] = add i32 [[B17]], [[S2]]
-; FILTER-NEXT: call void @foo(i32 [[U17]])
-; FILTER-NEXT: call void @foo(i32 [[B1]])
-; FILTER-NEXT: call void @foo(i32 [[B2]])
-; FILTER-NEXT: call void @foo(i32 [[B3]])
-; FILTER-NEXT: call void @foo(i32 [[B4]])
-; FILTER-NEXT: call void @foo(i32 [[B5]])
-; FILTER-NEXT: call void @foo(i32 [[B6]])
-; FILTER-NEXT: call void @foo(i32 [[B7]])
-; FILTER-NEXT: call void @foo(i32 [[B8]])
-; FILTER-NEXT: call void @foo(i32 [[B9]])
-; FILTER-NEXT: call void @foo(i32 [[B10]])
-; FILTER-NEXT: call void @foo(i32 [[B11]])
-; FILTER-NEXT: call void @foo(i32 [[B12]])
-; FILTER-NEXT: call void @foo(i32 [[B13]])
-; FILTER-NEXT: call void @foo(i32 [[B14]])
-; FILTER-NEXT: call void @foo(i32 [[B15]])
-; FILTER-NEXT: call void @foo(i32 [[B16]])
-; FILTER-NEXT: call void @foo(i32 [[B17]])
-; FILTER-NEXT: ret void
-;
-; OFF-LABEL: define void @many_bases_overlapping(
-; OFF-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; OFF-NEXT: [[ENTRY:.*:]]
-; OFF-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T1]])
-; OFF-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T2]])
-; OFF-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T3]])
-; OFF-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T4]])
-; OFF-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T5]])
-; OFF-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T6]])
-; OFF-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T7]])
-; OFF-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T8]])
-; OFF-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T9]])
-; OFF-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T10]])
-; OFF-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T11]])
-; OFF-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T12]])
-; OFF-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T13]])
-; OFF-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T14]])
-; OFF-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T15]])
-; OFF-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T16]])
-; OFF-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[T17]])
-; OFF-NEXT: [[U1:%.*]] = add i32 [[T1]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U1]])
-; OFF-NEXT: [[U2:%.*]] = add i32 [[T2]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U2]])
-; OFF-NEXT: [[U3:%.*]] = add i32 [[T3]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U3]])
-; OFF-NEXT: [[U4:%.*]] = add i32 [[T4]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U4]])
-; OFF-NEXT: [[U5:%.*]] = add i32 [[T5]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U5]])
-; OFF-NEXT: [[U6:%.*]] = add i32 [[T6]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U6]])
-; OFF-NEXT: [[U7:%.*]] = add i32 [[T7]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U7]])
-; OFF-NEXT: [[U8:%.*]] = add i32 [[T8]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U8]])
-; OFF-NEXT: [[U9:%.*]] = add i32 [[T9]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U9]])
-; OFF-NEXT: [[U10:%.*]] = add i32 [[T10]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U10]])
-; OFF-NEXT: [[U11:%.*]] = add i32 [[T11]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U11]])
-; OFF-NEXT: [[U12:%.*]] = add i32 [[T12]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U12]])
-; OFF-NEXT: [[U13:%.*]] = add i32 [[T13]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U13]])
-; OFF-NEXT: [[U14:%.*]] = add i32 [[T14]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U14]])
-; OFF-NEXT: [[U15:%.*]] = add i32 [[T15]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U15]])
-; OFF-NEXT: [[U16:%.*]] = add i32 [[T16]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U16]])
-; OFF-NEXT: [[U17:%.*]] = add i32 [[T17]], [[S]]
-; OFF-NEXT: call void @foo(i32 [[U17]])
-; OFF-NEXT: call void @foo(i32 [[B1]])
-; OFF-NEXT: call void @foo(i32 [[B2]])
-; OFF-NEXT: call void @foo(i32 [[B3]])
-; OFF-NEXT: call void @foo(i32 [[B4]])
-; OFF-NEXT: call void @foo(i32 [[B5]])
-; OFF-NEXT: call void @foo(i32 [[B6]])
-; OFF-NEXT: call void @foo(i32 [[B7]])
-; OFF-NEXT: call void @foo(i32 [[B8]])
-; OFF-NEXT: call void @foo(i32 [[B9]])
-; OFF-NEXT: call void @foo(i32 [[B10]])
-; OFF-NEXT: call void @foo(i32 [[B11]])
-; OFF-NEXT: call void @foo(i32 [[B12]])
-; OFF-NEXT: call void @foo(i32 [[B13]])
-; OFF-NEXT: call void @foo(i32 [[B14]])
-; OFF-NEXT: call void @foo(i32 [[B15]])
-; OFF-NEXT: call void @foo(i32 [[B16]])
-; OFF-NEXT: call void @foo(i32 [[B17]])
-; OFF-NEXT: ret void
-;
-; BUDGET32-LABEL: define void @many_bases_overlapping(
-; BUDGET32-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; BUDGET32-NEXT: [[ENTRY:.*:]]
-; BUDGET32-NEXT: [[S2:%.*]] = shl i32 [[S]], 1
-; BUDGET32-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T1]])
-; BUDGET32-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T2]])
-; BUDGET32-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T3]])
-; BUDGET32-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T4]])
-; BUDGET32-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T5]])
-; BUDGET32-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T6]])
-; BUDGET32-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T7]])
-; BUDGET32-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T8]])
-; BUDGET32-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T9]])
-; BUDGET32-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T10]])
-; BUDGET32-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T11]])
-; BUDGET32-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T12]])
-; BUDGET32-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T13]])
-; BUDGET32-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T14]])
-; BUDGET32-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T15]])
-; BUDGET32-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T16]])
-; BUDGET32-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
-; BUDGET32-NEXT: call void @foo(i32 [[T17]])
-; BUDGET32-NEXT: [[U1:%.*]] = add i32 [[B1]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U1]])
-; BUDGET32-NEXT: [[U2:%.*]] = add i32 [[B2]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U2]])
-; BUDGET32-NEXT: [[U3:%.*]] = add i32 [[B3]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U3]])
-; BUDGET32-NEXT: [[U4:%.*]] = add i32 [[B4]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U4]])
-; BUDGET32-NEXT: [[U5:%.*]] = add i32 [[B5]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U5]])
-; BUDGET32-NEXT: [[U6:%.*]] = add i32 [[B6]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U6]])
-; BUDGET32-NEXT: [[U7:%.*]] = add i32 [[B7]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U7]])
-; BUDGET32-NEXT: [[U8:%.*]] = add i32 [[B8]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U8]])
-; BUDGET32-NEXT: [[U9:%.*]] = add i32 [[B9]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U9]])
-; BUDGET32-NEXT: [[U10:%.*]] = add i32 [[B10]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U10]])
-; BUDGET32-NEXT: [[U11:%.*]] = add i32 [[B11]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U11]])
-; BUDGET32-NEXT: [[U12:%.*]] = add i32 [[B12]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U12]])
-; BUDGET32-NEXT: [[U13:%.*]] = add i32 [[B13]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U13]])
-; BUDGET32-NEXT: [[U14:%.*]] = add i32 [[B14]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U14]])
-; BUDGET32-NEXT: [[U15:%.*]] = add i32 [[B15]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U15]])
-; BUDGET32-NEXT: [[U16:%.*]] = add i32 [[B16]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U16]])
-; BUDGET32-NEXT: [[U17:%.*]] = add i32 [[B17]], [[S2]]
-; BUDGET32-NEXT: call void @foo(i32 [[U17]])
-; BUDGET32-NEXT: call void @foo(i32 [[B1]])
-; BUDGET32-NEXT: call void @foo(i32 [[B2]])
-; BUDGET32-NEXT: call void @foo(i32 [[B3]])
-; BUDGET32-NEXT: call void @foo(i32 [[B4]])
-; BUDGET32-NEXT: call void @foo(i32 [[B5]])
-; BUDGET32-NEXT: call void @foo(i32 [[B6]])
-; BUDGET32-NEXT: call void @foo(i32 [[B7]])
-; BUDGET32-NEXT: call void @foo(i32 [[B8]])
-; BUDGET32-NEXT: call void @foo(i32 [[B9]])
-; BUDGET32-NEXT: call void @foo(i32 [[B10]])
-; BUDGET32-NEXT: call void @foo(i32 [[B11]])
-; BUDGET32-NEXT: call void @foo(i32 [[B12]])
-; BUDGET32-NEXT: call void @foo(i32 [[B13]])
-; BUDGET32-NEXT: call void @foo(i32 [[B14]])
-; BUDGET32-NEXT: call void @foo(i32 [[B15]])
-; BUDGET32-NEXT: call void @foo(i32 [[B16]])
-; BUDGET32-NEXT: call void @foo(i32 [[B17]])
-; BUDGET32-NEXT: ret void
-;
-; ATTRS-LABEL: define void @many_bases_overlapping(
-; ATTRS-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; ATTRS-NEXT: [[ENTRY:.*:]]
-; ATTRS-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T1]])
-; ATTRS-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T2]])
-; ATTRS-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T3]])
-; ATTRS-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T4]])
-; ATTRS-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T5]])
-; ATTRS-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T6]])
-; ATTRS-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T7]])
-; ATTRS-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T8]])
-; ATTRS-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T9]])
-; ATTRS-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T10]])
-; ATTRS-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T11]])
-; ATTRS-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T12]])
-; ATTRS-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T13]])
-; ATTRS-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T14]])
-; ATTRS-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T15]])
-; ATTRS-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T16]])
-; ATTRS-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[T17]])
-; ATTRS-NEXT: [[U1:%.*]] = add i32 [[T1]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U1]])
-; ATTRS-NEXT: [[U2:%.*]] = add i32 [[T2]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U2]])
-; ATTRS-NEXT: [[U3:%.*]] = add i32 [[T3]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U3]])
-; ATTRS-NEXT: [[U4:%.*]] = add i32 [[T4]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U4]])
-; ATTRS-NEXT: [[U5:%.*]] = add i32 [[T5]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U5]])
-; ATTRS-NEXT: [[U6:%.*]] = add i32 [[T6]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U6]])
-; ATTRS-NEXT: [[U7:%.*]] = add i32 [[T7]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U7]])
-; ATTRS-NEXT: [[U8:%.*]] = add i32 [[T8]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U8]])
-; ATTRS-NEXT: [[U9:%.*]] = add i32 [[T9]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U9]])
-; ATTRS-NEXT: [[U10:%.*]] = add i32 [[T10]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U10]])
-; ATTRS-NEXT: [[U11:%.*]] = add i32 [[T11]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U11]])
-; ATTRS-NEXT: [[U12:%.*]] = add i32 [[T12]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U12]])
-; ATTRS-NEXT: [[U13:%.*]] = add i32 [[T13]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U13]])
-; ATTRS-NEXT: [[U14:%.*]] = add i32 [[T14]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U14]])
-; ATTRS-NEXT: [[U15:%.*]] = add i32 [[T15]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U15]])
-; ATTRS-NEXT: [[U16:%.*]] = add i32 [[T16]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U16]])
-; ATTRS-NEXT: [[U17:%.*]] = add i32 [[T17]], [[S]]
-; ATTRS-NEXT: call void @foo(i32 [[U17]])
-; ATTRS-NEXT: call void @foo(i32 [[B1]])
-; ATTRS-NEXT: call void @foo(i32 [[B2]])
-; ATTRS-NEXT: call void @foo(i32 [[B3]])
-; ATTRS-NEXT: call void @foo(i32 [[B4]])
-; ATTRS-NEXT: call void @foo(i32 [[B5]])
-; ATTRS-NEXT: call void @foo(i32 [[B6]])
-; ATTRS-NEXT: call void @foo(i32 [[B7]])
-; ATTRS-NEXT: call void @foo(i32 [[B8]])
-; ATTRS-NEXT: call void @foo(i32 [[B9]])
-; ATTRS-NEXT: call void @foo(i32 [[B10]])
-; ATTRS-NEXT: call void @foo(i32 [[B11]])
-; ATTRS-NEXT: call void @foo(i32 [[B12]])
-; ATTRS-NEXT: call void @foo(i32 [[B13]])
-; ATTRS-NEXT: call void @foo(i32 [[B14]])
-; ATTRS-NEXT: call void @foo(i32 [[B15]])
-; ATTRS-NEXT: call void @foo(i32 [[B16]])
-; ATTRS-NEXT: call void @foo(i32 [[B17]])
-; ATTRS-NEXT: ret void
-;
-; NOFILTER-LABEL: define void @many_bases_overlapping(
-; NOFILTER-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; NOFILTER-NEXT: [[ENTRY:.*:]]
-; NOFILTER-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T1]])
-; NOFILTER-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T2]])
-; NOFILTER-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T3]])
-; NOFILTER-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T4]])
-; NOFILTER-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T5]])
-; NOFILTER-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T6]])
-; NOFILTER-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T7]])
-; NOFILTER-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T8]])
-; NOFILTER-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T9]])
-; NOFILTER-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T10]])
-; NOFILTER-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T11]])
-; NOFILTER-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T12]])
-; NOFILTER-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T13]])
-; NOFILTER-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T14]])
-; NOFILTER-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T15]])
-; NOFILTER-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T16]])
-; NOFILTER-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[T17]])
-; NOFILTER-NEXT: [[U1:%.*]] = add i32 [[T1]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U1]])
-; NOFILTER-NEXT: [[U2:%.*]] = add i32 [[T2]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U2]])
-; NOFILTER-NEXT: [[U3:%.*]] = add i32 [[T3]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U3]])
-; NOFILTER-NEXT: [[U4:%.*]] = add i32 [[T4]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U4]])
-; NOFILTER-NEXT: [[U5:%.*]] = add i32 [[T5]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U5]])
-; NOFILTER-NEXT: [[U6:%.*]] = add i32 [[T6]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U6]])
-; NOFILTER-NEXT: [[U7:%.*]] = add i32 [[T7]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U7]])
-; NOFILTER-NEXT: [[U8:%.*]] = add i32 [[T8]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U8]])
-; NOFILTER-NEXT: [[U9:%.*]] = add i32 [[T9]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U9]])
-; NOFILTER-NEXT: [[U10:%.*]] = add i32 [[T10]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U10]])
-; NOFILTER-NEXT: [[U11:%.*]] = add i32 [[T11]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U11]])
-; NOFILTER-NEXT: [[U12:%.*]] = add i32 [[T12]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U12]])
-; NOFILTER-NEXT: [[U13:%.*]] = add i32 [[T13]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U13]])
-; NOFILTER-NEXT: [[U14:%.*]] = add i32 [[T14]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U14]])
-; NOFILTER-NEXT: [[U15:%.*]] = add i32 [[T15]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U15]])
-; NOFILTER-NEXT: [[U16:%.*]] = add i32 [[T16]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U16]])
-; NOFILTER-NEXT: [[U17:%.*]] = add i32 [[T17]], [[S]]
-; NOFILTER-NEXT: call void @foo(i32 [[U17]])
-; NOFILTER-NEXT: call void @foo(i32 [[B1]])
-; NOFILTER-NEXT: call void @foo(i32 [[B2]])
-; NOFILTER-NEXT: call void @foo(i32 [[B3]])
-; NOFILTER-NEXT: call void @foo(i32 [[B4]])
-; NOFILTER-NEXT: call void @foo(i32 [[B5]])
-; NOFILTER-NEXT: call void @foo(i32 [[B6]])
-; NOFILTER-NEXT: call void @foo(i32 [[B7]])
-; NOFILTER-NEXT: call void @foo(i32 [[B8]])
-; NOFILTER-NEXT: call void @foo(i32 [[B9]])
-; NOFILTER-NEXT: call void @foo(i32 [[B10]])
-; NOFILTER-NEXT: call void @foo(i32 [[B11]])
-; NOFILTER-NEXT: call void @foo(i32 [[B12]])
-; NOFILTER-NEXT: call void @foo(i32 [[B13]])
-; NOFILTER-NEXT: call void @foo(i32 [[B14]])
-; NOFILTER-NEXT: call void @foo(i32 [[B15]])
-; NOFILTER-NEXT: call void @foo(i32 [[B16]])
-; NOFILTER-NEXT: call void @foo(i32 [[B17]])
-; NOFILTER-NEXT: ret void
-;
-entry:
- %s2 = shl i32 %s, 1
- %t1 = add i32 %b1, %s
- call void @foo(i32 %t1)
- %t2 = add i32 %b2, %s
- call void @foo(i32 %t2)
- %t3 = add i32 %b3, %s
- call void @foo(i32 %t3)
- %t4 = add i32 %b4, %s
- call void @foo(i32 %t4)
- %t5 = add i32 %b5, %s
- call void @foo(i32 %t5)
- %t6 = add i32 %b6, %s
- call void @foo(i32 %t6)
- %t7 = add i32 %b7, %s
- call void @foo(i32 %t7)
- %t8 = add i32 %b8, %s
- call void @foo(i32 %t8)
- %t9 = add i32 %b9, %s
- call void @foo(i32 %t9)
- %t10 = add i32 %b10, %s
- call void @foo(i32 %t10)
- %t11 = add i32 %b11, %s
- call void @foo(i32 %t11)
- %t12 = add i32 %b12, %s
- call void @foo(i32 %t12)
- %t13 = add i32 %b13, %s
- call void @foo(i32 %t13)
- %t14 = add i32 %b14, %s
- call void @foo(i32 %t14)
- %t15 = add i32 %b15, %s
- call void @foo(i32 %t15)
- %t16 = add i32 %b16, %s
- call void @foo(i32 %t16)
- %t17 = add i32 %b17, %s
- call void @foo(i32 %t17)
- %u1 = add i32 %b1, %s2
- call void @foo(i32 %u1)
- %u2 = add i32 %b2, %s2
- call void @foo(i32 %u2)
- %u3 = add i32 %b3, %s2
- call void @foo(i32 %u3)
- %u4 = add i32 %b4, %s2
- call void @foo(i32 %u4)
- %u5 = add i32 %b5, %s2
- call void @foo(i32 %u5)
- %u6 = add i32 %b6, %s2
- call void @foo(i32 %u6)
- %u7 = add i32 %b7, %s2
- call void @foo(i32 %u7)
- %u8 = add i32 %b8, %s2
- call void @foo(i32 %u8)
- %u9 = add i32 %b9, %s2
- call void @foo(i32 %u9)
- %u10 = add i32 %b10, %s2
- call void @foo(i32 %u10)
- %u11 = add i32 %b11, %s2
- call void @foo(i32 %u11)
- %u12 = add i32 %b12, %s2
- call void @foo(i32 %u12)
- %u13 = add i32 %b13, %s2
- call void @foo(i32 %u13)
- %u14 = add i32 %b14, %s2
- call void @foo(i32 %u14)
- %u15 = add i32 %b15, %s2
- call void @foo(i32 %u15)
- %u16 = add i32 %b16, %s2
- call void @foo(i32 %u16)
- %u17 = add i32 %b17, %s2
- call void @foo(i32 %u17)
- call void @foo(i32 %b1)
- call void @foo(i32 %b2)
- call void @foo(i32 %b3)
- call void @foo(i32 %b4)
- call void @foo(i32 %b5)
- call void @foo(i32 %b6)
- call void @foo(i32 %b7)
- call void @foo(i32 %b8)
- call void @foo(i32 %b9)
- call void @foo(i32 %b10)
- call void @foo(i32 %b11)
- call void @foo(i32 %b12)
- call void @foo(i32 %b13)
- call void @foo(i32 %b14)
- call void @foo(i32 %b15)
- call void @foo(i32 %b16)
- call void @foo(i32 %b17)
- ret void
-}
-
-define void @many_bases_adjacent(i32 %s, i32 %b1, i32 %b2, i32 %b3, i32 %b4, i32 %b5, i32 %b6, i32 %b7, i32 %b8, i32 %b9, i32 %b10, i32 %b11, i32 %b12, i32 %b13, i32 %b14, i32 %b15, i32 %b16, i32 %b17) {
-; CHECK-LABEL: define void @many_bases_adjacent(
-; CHECK-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T1]])
-; CHECK-NEXT: [[U1:%.*]] = add i32 [[T1]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U1]])
-; CHECK-NEXT: call void @foo(i32 [[B1]])
-; CHECK-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T2]])
-; CHECK-NEXT: [[U2:%.*]] = add i32 [[T2]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U2]])
-; CHECK-NEXT: call void @foo(i32 [[B2]])
-; CHECK-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T3]])
-; CHECK-NEXT: [[U3:%.*]] = add i32 [[T3]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U3]])
-; CHECK-NEXT: call void @foo(i32 [[B3]])
-; CHECK-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T4]])
-; CHECK-NEXT: [[U4:%.*]] = add i32 [[T4]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U4]])
-; CHECK-NEXT: call void @foo(i32 [[B4]])
-; CHECK-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T5]])
-; CHECK-NEXT: [[U5:%.*]] = add i32 [[T5]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U5]])
-; CHECK-NEXT: call void @foo(i32 [[B5]])
-; CHECK-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T6]])
-; CHECK-NEXT: [[U6:%.*]] = add i32 [[T6]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U6]])
-; CHECK-NEXT: call void @foo(i32 [[B6]])
-; CHECK-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T7]])
-; CHECK-NEXT: [[U7:%.*]] = add i32 [[T7]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U7]])
-; CHECK-NEXT: call void @foo(i32 [[B7]])
-; CHECK-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T8]])
-; CHECK-NEXT: [[U8:%.*]] = add i32 [[T8]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U8]])
-; CHECK-NEXT: call void @foo(i32 [[B8]])
-; CHECK-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T9]])
-; CHECK-NEXT: [[U9:%.*]] = add i32 [[T9]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U9]])
-; CHECK-NEXT: call void @foo(i32 [[B9]])
-; CHECK-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T10]])
-; CHECK-NEXT: [[U10:%.*]] = add i32 [[T10]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U10]])
-; CHECK-NEXT: call void @foo(i32 [[B10]])
-; CHECK-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T11]])
-; CHECK-NEXT: [[U11:%.*]] = add i32 [[T11]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U11]])
-; CHECK-NEXT: call void @foo(i32 [[B11]])
-; CHECK-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T12]])
-; CHECK-NEXT: [[U12:%.*]] = add i32 [[T12]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U12]])
-; CHECK-NEXT: call void @foo(i32 [[B12]])
-; CHECK-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T13]])
-; CHECK-NEXT: [[U13:%.*]] = add i32 [[T13]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U13]])
-; CHECK-NEXT: call void @foo(i32 [[B13]])
-; CHECK-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T14]])
-; CHECK-NEXT: [[U14:%.*]] = add i32 [[T14]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U14]])
-; CHECK-NEXT: call void @foo(i32 [[B14]])
-; CHECK-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T15]])
-; CHECK-NEXT: [[U15:%.*]] = add i32 [[T15]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U15]])
-; CHECK-NEXT: call void @foo(i32 [[B15]])
-; CHECK-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T16]])
-; CHECK-NEXT: [[U16:%.*]] = add i32 [[T16]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U16]])
-; CHECK-NEXT: call void @foo(i32 [[B16]])
-; CHECK-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[T17]])
-; CHECK-NEXT: [[U17:%.*]] = add i32 [[T17]], [[S]]
-; CHECK-NEXT: call void @foo(i32 [[U17]])
-; CHECK-NEXT: call void @foo(i32 [[B17]])
-; CHECK-NEXT: ret void
-;
-entry:
- %s2 = shl i32 %s, 1
- %t1 = add i32 %b1, %s
- call void @foo(i32 %t1)
- %u1 = add i32 %b1, %s2
- call void @foo(i32 %u1)
- call void @foo(i32 %b1)
- %t2 = add i32 %b2, %s
- call void @foo(i32 %t2)
- %u2 = add i32 %b2, %s2
- call void @foo(i32 %u2)
- call void @foo(i32 %b2)
- %t3 = add i32 %b3, %s
- call void @foo(i32 %t3)
- %u3 = add i32 %b3, %s2
- call void @foo(i32 %u3)
- call void @foo(i32 %b3)
- %t4 = add i32 %b4, %s
- call void @foo(i32 %t4)
- %u4 = add i32 %b4, %s2
- call void @foo(i32 %u4)
- call void @foo(i32 %b4)
- %t5 = add i32 %b5, %s
- call void @foo(i32 %t5)
- %u5 = add i32 %b5, %s2
- call void @foo(i32 %u5)
- call void @foo(i32 %b5)
- %t6 = add i32 %b6, %s
- call void @foo(i32 %t6)
- %u6 = add i32 %b6, %s2
- call void @foo(i32 %u6)
- call void @foo(i32 %b6)
- %t7 = add i32 %b7, %s
- call void @foo(i32 %t7)
- %u7 = add i32 %b7, %s2
- call void @foo(i32 %u7)
- call void @foo(i32 %b7)
- %t8 = add i32 %b8, %s
- call void @foo(i32 %t8)
- %u8 = add i32 %b8, %s2
- call void @foo(i32 %u8)
- call void @foo(i32 %b8)
- %t9 = add i32 %b9, %s
- call void @foo(i32 %t9)
- %u9 = add i32 %b9, %s2
- call void @foo(i32 %u9)
- call void @foo(i32 %b9)
- %t10 = add i32 %b10, %s
- call void @foo(i32 %t10)
- %u10 = add i32 %b10, %s2
- call void @foo(i32 %u10)
- call void @foo(i32 %b10)
- %t11 = add i32 %b11, %s
- call void @foo(i32 %t11)
- %u11 = add i32 %b11, %s2
- call void @foo(i32 %u11)
- call void @foo(i32 %b11)
- %t12 = add i32 %b12, %s
- call void @foo(i32 %t12)
- %u12 = add i32 %b12, %s2
- call void @foo(i32 %u12)
- call void @foo(i32 %b12)
- %t13 = add i32 %b13, %s
- call void @foo(i32 %t13)
- %u13 = add i32 %b13, %s2
- call void @foo(i32 %u13)
- call void @foo(i32 %b13)
- %t14 = add i32 %b14, %s
- call void @foo(i32 %t14)
- %u14 = add i32 %b14, %s2
- call void @foo(i32 %u14)
- call void @foo(i32 %b14)
- %t15 = add i32 %b15, %s
- call void @foo(i32 %t15)
- %u15 = add i32 %b15, %s2
- call void @foo(i32 %u15)
- call void @foo(i32 %b15)
- %t16 = add i32 %b16, %s
- call void @foo(i32 %t16)
- %u16 = add i32 %b16, %s2
- call void @foo(i32 %u16)
- call void @foo(i32 %b16)
- %t17 = add i32 %b17, %s
- call void @foo(i32 %t17)
- %u17 = add i32 %b17, %s2
- call void @foo(i32 %u17)
- call void @foo(i32 %b17)
- ret void
-}
-
-define void @four_bases(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4) {
-; CHECK-LABEL: define void @four_bases(
-; CHECK-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]]) {
+define void @peak_above_budget(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) {
+; CHECK-LABEL: define void @peak_above_budget(
+; CHECK-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[S2:%.*]] = shl i128 [[S]], 1
; CHECK-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
; CHECK-NEXT: call void @bar(i128 [[T1]])
; CHECK-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
@@ -780,324 +41,85 @@ define void @four_bases(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4) {
; CHECK-NEXT: call void @bar(i128 [[T3]])
; CHECK-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
; CHECK-NEXT: call void @bar(i128 [[T4]])
-; CHECK-NEXT: [[U1:%.*]] = add i128 [[T1]], [[S]]
+; CHECK-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T5]])
+; CHECK-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T6]])
+; CHECK-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T7]])
+; CHECK-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T8]])
+; CHECK-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T9]])
+; CHECK-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T10]])
+; CHECK-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T11]])
+; CHECK-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T12]])
+; CHECK-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T13]])
+; CHECK-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T14]])
+; CHECK-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T15]])
+; CHECK-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T16]])
+; CHECK-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T17]])
+; CHECK-NEXT: [[U1:%.*]] = add i128 [[B1]], [[S2]]
; CHECK-NEXT: call void @bar(i128 [[U1]])
-; CHECK-NEXT: [[U2:%.*]] = add i128 [[T2]], [[S]]
+; CHECK-NEXT: [[U2:%.*]] = add i128 [[B2]], [[S2]]
; CHECK-NEXT: call void @bar(i128 [[U2]])
-; CHECK-NEXT: [[U3:%.*]] = add i128 [[T3]], [[S]]
+; CHECK-NEXT: [[U3:%.*]] = add i128 [[B3]], [[S2]]
; CHECK-NEXT: call void @bar(i128 [[U3]])
-; CHECK-NEXT: [[U4:%.*]] = add i128 [[T4]], [[S]]
+; CHECK-NEXT: [[U4:%.*]] = add i128 [[B4]], [[S2]]
; CHECK-NEXT: call void @bar(i128 [[U4]])
+; CHECK-NEXT: [[U5:%.*]] = add i128 [[B5]], [[S2]]
+; CHECK-NEXT: call void @bar(i128 [[U5]])
+; CHECK-NEXT: [[U6:%.*]] = add i128 [[B6]], [[S2]]
+; CHECK-NEXT: call void @bar(i128 [[U6]])
+; CHECK-NEXT: [[U7:%.*]] = add i128 [[B7]], [[S2]]
+; CHECK-NEXT: call void @bar(i128 [[U7]])
+; CHECK-NEXT: [[U8:%.*]] = add i128 [[B8]], [[S2]]
+; CHECK-NEXT: call void @bar(i128 [[U8]])
+; CHECK-NEXT: [[U9:%.*]] = add i128 [[B9]], [[S2]]
+; CHECK-NEXT: call void @bar(i128 [[U9]])
+; CHECK-NEXT: [[U10:%.*]] = add i128 [[B10]], [[S2]]
+; CHECK-NEXT: call void @bar(i128 [[U10]])
+; CHECK-NEXT: [[U11:%.*]] = add i128 [[B11]], [[S2]]
+; CHECK-NEXT: call void @bar(i128 [[U11]])
+; CHECK-NEXT: [[U12:%.*]] = add i128 [[B12]], [[S2]]
+; CHECK-NEXT: call void @bar(i128 [[U12]])
+; CHECK-NEXT: [[U13:%.*]] = add i128 [[B13]], [[S2]]
+; CHECK-NEXT: call void @bar(i128 [[U13]])
+; CHECK-NEXT: [[U14:%.*]] = add i128 [[B14]], [[S2]]
+; CHECK-NEXT: call void @bar(i128 [[U14]])
+; CHECK-NEXT: [[U15:%.*]] = add i128 [[B15]], [[S2]]
+; CHECK-NEXT: call void @bar(i128 [[U15]])
+; CHECK-NEXT: [[U16:%.*]] = add i128 [[B16]], [[S2]]
+; CHECK-NEXT: call void @bar(i128 [[U16]])
+; CHECK-NEXT: [[U17:%.*]] = add i128 [[B17]], [[S2]]
+; CHECK-NEXT: call void @bar(i128 [[U17]])
; CHECK-NEXT: call void @bar(i128 [[B1]])
; CHECK-NEXT: call void @bar(i128 [[B2]])
; CHECK-NEXT: call void @bar(i128 [[B3]])
; CHECK-NEXT: call void @bar(i128 [[B4]])
+; CHECK-NEXT: call void @bar(i128 [[B5]])
+; CHECK-NEXT: call void @bar(i128 [[B6]])
+; CHECK-NEXT: call void @bar(i128 [[B7]])
+; CHECK-NEXT: call void @bar(i128 [[B8]])
+; CHECK-NEXT: call void @bar(i128 [[B9]])
+; CHECK-NEXT: call void @bar(i128 [[B10]])
+; CHECK-NEXT: call void @bar(i128 [[B11]])
+; CHECK-NEXT: call void @bar(i128 [[B12]])
+; CHECK-NEXT: call void @bar(i128 [[B13]])
+; CHECK-NEXT: call void @bar(i128 [[B14]])
+; CHECK-NEXT: call void @bar(i128 [[B15]])
+; CHECK-NEXT: call void @bar(i128 [[B16]])
+; CHECK-NEXT: call void @bar(i128 [[B17]])
; CHECK-NEXT: ret void
;
-entry:
- %s2 = shl i128 %s, 1
- %t1 = add i128 %b1, %s
- call void @bar(i128 %t1)
- %t2 = add i128 %b2, %s
- call void @bar(i128 %t2)
- %t3 = add i128 %b3, %s
- call void @bar(i128 %t3)
- %t4 = add i128 %b4, %s
- call void @bar(i128 %t4)
- %u1 = add i128 %b1, %s2
- call void @bar(i128 %u1)
- %u2 = add i128 %b2, %s2
- call void @bar(i128 %u2)
- %u3 = add i128 %b3, %s2
- call void @bar(i128 %u3)
- %u4 = add i128 %b4, %s2
- call void @bar(i128 %u4)
- call void @bar(i128 %b1)
- call void @bar(i128 %b2)
- call void @bar(i128 %b3)
- call void @bar(i128 %b4)
- ret void
-}
-
-; i128 values weigh 4 registers each, so these three bodies peak at 144 with the
-; rewrites applied against 80 without. That straddles the budgets implied by the
-; occupancy attributes, letting the real TTI budget decide with no override.
-
-; No attributes: flat work group size defaults to 1024, so 4 waves per EU and a
-; 128-register budget. 144 exceeds the 115-register safe budget, so filtered.
-define void @default_occupancy(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) {
-; BUDGET32-LABEL: define void @default_occupancy(
-; BUDGET32-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) {
-; BUDGET32-NEXT: [[ENTRY:.*:]]
-; BUDGET32-NEXT: [[S2:%.*]] = shl i128 [[S]], 1
-; BUDGET32-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T1]])
-; BUDGET32-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T2]])
-; BUDGET32-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T3]])
-; BUDGET32-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T4]])
-; BUDGET32-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T5]])
-; BUDGET32-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T6]])
-; BUDGET32-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T7]])
-; BUDGET32-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T8]])
-; BUDGET32-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T9]])
-; BUDGET32-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T10]])
-; BUDGET32-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T11]])
-; BUDGET32-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T12]])
-; BUDGET32-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T13]])
-; BUDGET32-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T14]])
-; BUDGET32-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T15]])
-; BUDGET32-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T16]])
-; BUDGET32-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T17]])
-; BUDGET32-NEXT: [[U1:%.*]] = add i128 [[B1]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U1]])
-; BUDGET32-NEXT: [[U2:%.*]] = add i128 [[B2]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U2]])
-; BUDGET32-NEXT: [[U3:%.*]] = add i128 [[B3]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U3]])
-; BUDGET32-NEXT: [[U4:%.*]] = add i128 [[B4]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U4]])
-; BUDGET32-NEXT: [[U5:%.*]] = add i128 [[B5]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U5]])
-; BUDGET32-NEXT: [[U6:%.*]] = add i128 [[B6]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U6]])
-; BUDGET32-NEXT: [[U7:%.*]] = add i128 [[B7]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U7]])
-; BUDGET32-NEXT: [[U8:%.*]] = add i128 [[B8]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U8]])
-; BUDGET32-NEXT: [[U9:%.*]] = add i128 [[B9]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U9]])
-; BUDGET32-NEXT: [[U10:%.*]] = add i128 [[B10]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U10]])
-; BUDGET32-NEXT: [[U11:%.*]] = add i128 [[B11]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U11]])
-; BUDGET32-NEXT: [[U12:%.*]] = add i128 [[B12]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U12]])
-; BUDGET32-NEXT: [[U13:%.*]] = add i128 [[B13]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U13]])
-; BUDGET32-NEXT: [[U14:%.*]] = add i128 [[B14]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U14]])
-; BUDGET32-NEXT: [[U15:%.*]] = add i128 [[B15]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U15]])
-; BUDGET32-NEXT: [[U16:%.*]] = add i128 [[B16]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U16]])
-; BUDGET32-NEXT: [[U17:%.*]] = add i128 [[B17]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U17]])
-; BUDGET32-NEXT: call void @bar(i128 [[B1]])
-; BUDGET32-NEXT: call void @bar(i128 [[B2]])
-; BUDGET32-NEXT: call void @bar(i128 [[B3]])
-; BUDGET32-NEXT: call void @bar(i128 [[B4]])
-; BUDGET32-NEXT: call void @bar(i128 [[B5]])
-; BUDGET32-NEXT: call void @bar(i128 [[B6]])
-; BUDGET32-NEXT: call void @bar(i128 [[B7]])
-; BUDGET32-NEXT: call void @bar(i128 [[B8]])
-; BUDGET32-NEXT: call void @bar(i128 [[B9]])
-; BUDGET32-NEXT: call void @bar(i128 [[B10]])
-; BUDGET32-NEXT: call void @bar(i128 [[B11]])
-; BUDGET32-NEXT: call void @bar(i128 [[B12]])
-; BUDGET32-NEXT: call void @bar(i128 [[B13]])
-; BUDGET32-NEXT: call void @bar(i128 [[B14]])
-; BUDGET32-NEXT: call void @bar(i128 [[B15]])
-; BUDGET32-NEXT: call void @bar(i128 [[B16]])
-; BUDGET32-NEXT: call void @bar(i128 [[B17]])
-; BUDGET32-NEXT: ret void
-;
-; ATTRS-LABEL: define void @default_occupancy(
-; ATTRS-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) {
-; ATTRS-NEXT: [[ENTRY:.*:]]
-; ATTRS-NEXT: [[S2:%.*]] = shl i128 [[S]], 1
-; ATTRS-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T1]])
-; ATTRS-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T2]])
-; ATTRS-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T3]])
-; ATTRS-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T4]])
-; ATTRS-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T5]])
-; ATTRS-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T6]])
-; ATTRS-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T7]])
-; ATTRS-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T8]])
-; ATTRS-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T9]])
-; ATTRS-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T10]])
-; ATTRS-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T11]])
-; ATTRS-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T12]])
-; ATTRS-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T13]])
-; ATTRS-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T14]])
-; ATTRS-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T15]])
-; ATTRS-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T16]])
-; ATTRS-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T17]])
-; ATTRS-NEXT: [[U1:%.*]] = add i128 [[B1]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U1]])
-; ATTRS-NEXT: [[U2:%.*]] = add i128 [[B2]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U2]])
-; ATTRS-NEXT: [[U3:%.*]] = add i128 [[B3]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U3]])
-; ATTRS-NEXT: [[U4:%.*]] = add i128 [[B4]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U4]])
-; ATTRS-NEXT: [[U5:%.*]] = add i128 [[B5]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U5]])
-; ATTRS-NEXT: [[U6:%.*]] = add i128 [[B6]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U6]])
-; ATTRS-NEXT: [[U7:%.*]] = add i128 [[B7]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U7]])
-; ATTRS-NEXT: [[U8:%.*]] = add i128 [[B8]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U8]])
-; ATTRS-NEXT: [[U9:%.*]] = add i128 [[B9]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U9]])
-; ATTRS-NEXT: [[U10:%.*]] = add i128 [[B10]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U10]])
-; ATTRS-NEXT: [[U11:%.*]] = add i128 [[B11]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U11]])
-; ATTRS-NEXT: [[U12:%.*]] = add i128 [[B12]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U12]])
-; ATTRS-NEXT: [[U13:%.*]] = add i128 [[B13]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U13]])
-; ATTRS-NEXT: [[U14:%.*]] = add i128 [[B14]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U14]])
-; ATTRS-NEXT: [[U15:%.*]] = add i128 [[B15]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U15]])
-; ATTRS-NEXT: [[U16:%.*]] = add i128 [[B16]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U16]])
-; ATTRS-NEXT: [[U17:%.*]] = add i128 [[B17]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U17]])
-; ATTRS-NEXT: call void @bar(i128 [[B1]])
-; ATTRS-NEXT: call void @bar(i128 [[B2]])
-; ATTRS-NEXT: call void @bar(i128 [[B3]])
-; ATTRS-NEXT: call void @bar(i128 [[B4]])
-; ATTRS-NEXT: call void @bar(i128 [[B5]])
-; ATTRS-NEXT: call void @bar(i128 [[B6]])
-; ATTRS-NEXT: call void @bar(i128 [[B7]])
-; ATTRS-NEXT: call void @bar(i128 [[B8]])
-; ATTRS-NEXT: call void @bar(i128 [[B9]])
-; ATTRS-NEXT: call void @bar(i128 [[B10]])
-; ATTRS-NEXT: call void @bar(i128 [[B11]])
-; ATTRS-NEXT: call void @bar(i128 [[B12]])
-; ATTRS-NEXT: call void @bar(i128 [[B13]])
-; ATTRS-NEXT: call void @bar(i128 [[B14]])
-; ATTRS-NEXT: call void @bar(i128 [[B15]])
-; ATTRS-NEXT: call void @bar(i128 [[B16]])
-; ATTRS-NEXT: call void @bar(i128 [[B17]])
-; ATTRS-NEXT: ret void
-;
-; NOFILTER-LABEL: define void @default_occupancy(
-; NOFILTER-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) {
-; NOFILTER-NEXT: [[ENTRY:.*:]]
-; NOFILTER-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T1]])
-; NOFILTER-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T2]])
-; NOFILTER-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T3]])
-; NOFILTER-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T4]])
-; NOFILTER-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T5]])
-; NOFILTER-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T6]])
-; NOFILTER-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T7]])
-; NOFILTER-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T8]])
-; NOFILTER-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T9]])
-; NOFILTER-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T10]])
-; NOFILTER-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T11]])
-; NOFILTER-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T12]])
-; NOFILTER-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T13]])
-; NOFILTER-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T14]])
-; NOFILTER-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T15]])
-; NOFILTER-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T16]])
-; NOFILTER-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T17]])
-; NOFILTER-NEXT: [[U1:%.*]] = add i128 [[T1]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U1]])
-; NOFILTER-NEXT: [[U2:%.*]] = add i128 [[T2]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U2]])
-; NOFILTER-NEXT: [[U3:%.*]] = add i128 [[T3]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U3]])
-; NOFILTER-NEXT: [[U4:%.*]] = add i128 [[T4]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U4]])
-; NOFILTER-NEXT: [[U5:%.*]] = add i128 [[T5]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U5]])
-; NOFILTER-NEXT: [[U6:%.*]] = add i128 [[T6]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U6]])
-; NOFILTER-NEXT: [[U7:%.*]] = add i128 [[T7]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U7]])
-; NOFILTER-NEXT: [[U8:%.*]] = add i128 [[T8]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U8]])
-; NOFILTER-NEXT: [[U9:%.*]] = add i128 [[T9]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U9]])
-; NOFILTER-NEXT: [[U10:%.*]] = add i128 [[T10]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U10]])
-; NOFILTER-NEXT: [[U11:%.*]] = add i128 [[T11]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U11]])
-; NOFILTER-NEXT: [[U12:%.*]] = add i128 [[T12]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U12]])
-; NOFILTER-NEXT: [[U13:%.*]] = add i128 [[T13]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U13]])
-; NOFILTER-NEXT: [[U14:%.*]] = add i128 [[T14]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U14]])
-; NOFILTER-NEXT: [[U15:%.*]] = add i128 [[T15]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U15]])
-; NOFILTER-NEXT: [[U16:%.*]] = add i128 [[T16]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U16]])
-; NOFILTER-NEXT: [[U17:%.*]] = add i128 [[T17]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U17]])
-; NOFILTER-NEXT: call void @bar(i128 [[B1]])
-; NOFILTER-NEXT: call void @bar(i128 [[B2]])
-; NOFILTER-NEXT: call void @bar(i128 [[B3]])
-; NOFILTER-NEXT: call void @bar(i128 [[B4]])
-; NOFILTER-NEXT: call void @bar(i128 [[B5]])
-; NOFILTER-NEXT: call void @bar(i128 [[B6]])
-; NOFILTER-NEXT: call void @bar(i128 [[B7]])
-; NOFILTER-NEXT: call void @bar(i128 [[B8]])
-; NOFILTER-NEXT: call void @bar(i128 [[B9]])
-; NOFILTER-NEXT: call void @bar(i128 [[B10]])
-; NOFILTER-NEXT: call void @bar(i128 [[B11]])
-; NOFILTER-NEXT: call void @bar(i128 [[B12]])
-; NOFILTER-NEXT: call void @bar(i128 [[B13]])
-; NOFILTER-NEXT: call void @bar(i128 [[B14]])
-; NOFILTER-NEXT: call void @bar(i128 [[B15]])
-; NOFILTER-NEXT: call void @bar(i128 [[B16]])
-; NOFILTER-NEXT: call void @bar(i128 [[B17]])
-; NOFILTER-NEXT: ret void
-;
entry:
%s2 = shl i128 %s, 1
%t1 = add i128 %b1, %s
@@ -1188,645 +210,96 @@ entry:
ret void
}
-; A 256-thread work group is 4 wave64s over 4 EUs, so 1 wave per EU and the full
-; 512-register budget. The same 144 now fits, so the rewrites survive.
-define void @small_work_group(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) #0 {
-; BUDGET32-LABEL: define void @small_work_group(
-; BUDGET32-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR0:[0-9]+]] {
-; BUDGET32-NEXT: [[ENTRY:.*:]]
-; BUDGET32-NEXT: [[S2:%.*]] = shl i128 [[S]], 1
-; BUDGET32-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T1]])
-; BUDGET32-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T2]])
-; BUDGET32-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T3]])
-; BUDGET32-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T4]])
-; BUDGET32-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T5]])
-; BUDGET32-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T6]])
-; BUDGET32-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T7]])
-; BUDGET32-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T8]])
-; BUDGET32-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T9]])
-; BUDGET32-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T10]])
-; BUDGET32-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T11]])
-; BUDGET32-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T12]])
-; BUDGET32-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T13]])
-; BUDGET32-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T14]])
-; BUDGET32-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T15]])
-; BUDGET32-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T16]])
-; BUDGET32-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T17]])
-; BUDGET32-NEXT: [[U1:%.*]] = add i128 [[B1]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U1]])
-; BUDGET32-NEXT: [[U2:%.*]] = add i128 [[B2]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U2]])
-; BUDGET32-NEXT: [[U3:%.*]] = add i128 [[B3]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U3]])
-; BUDGET32-NEXT: [[U4:%.*]] = add i128 [[B4]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U4]])
-; BUDGET32-NEXT: [[U5:%.*]] = add i128 [[B5]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U5]])
-; BUDGET32-NEXT: [[U6:%.*]] = add i128 [[B6]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U6]])
-; BUDGET32-NEXT: [[U7:%.*]] = add i128 [[B7]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U7]])
-; BUDGET32-NEXT: [[U8:%.*]] = add i128 [[B8]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U8]])
-; BUDGET32-NEXT: [[U9:%.*]] = add i128 [[B9]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U9]])
-; BUDGET32-NEXT: [[U10:%.*]] = add i128 [[B10]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U10]])
-; BUDGET32-NEXT: [[U11:%.*]] = add i128 [[B11]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U11]])
-; BUDGET32-NEXT: [[U12:%.*]] = add i128 [[B12]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U12]])
-; BUDGET32-NEXT: [[U13:%.*]] = add i128 [[B13]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U13]])
-; BUDGET32-NEXT: [[U14:%.*]] = add i128 [[B14]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U14]])
-; BUDGET32-NEXT: [[U15:%.*]] = add i128 [[B15]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U15]])
-; BUDGET32-NEXT: [[U16:%.*]] = add i128 [[B16]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U16]])
-; BUDGET32-NEXT: [[U17:%.*]] = add i128 [[B17]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U17]])
-; BUDGET32-NEXT: call void @bar(i128 [[B1]])
-; BUDGET32-NEXT: call void @bar(i128 [[B2]])
-; BUDGET32-NEXT: call void @bar(i128 [[B3]])
-; BUDGET32-NEXT: call void @bar(i128 [[B4]])
-; BUDGET32-NEXT: call void @bar(i128 [[B5]])
-; BUDGET32-NEXT: call void @bar(i128 [[B6]])
-; BUDGET32-NEXT: call void @bar(i128 [[B7]])
-; BUDGET32-NEXT: call void @bar(i128 [[B8]])
-; BUDGET32-NEXT: call void @bar(i128 [[B9]])
-; BUDGET32-NEXT: call void @bar(i128 [[B10]])
-; BUDGET32-NEXT: call void @bar(i128 [[B11]])
-; BUDGET32-NEXT: call void @bar(i128 [[B12]])
-; BUDGET32-NEXT: call void @bar(i128 [[B13]])
-; BUDGET32-NEXT: call void @bar(i128 [[B14]])
-; BUDGET32-NEXT: call void @bar(i128 [[B15]])
-; BUDGET32-NEXT: call void @bar(i128 [[B16]])
-; BUDGET32-NEXT: call void @bar(i128 [[B17]])
-; BUDGET32-NEXT: ret void
-;
-; ATTRS-LABEL: define void @small_work_group(
-; ATTRS-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR0:[0-9]+]] {
-; ATTRS-NEXT: [[ENTRY:.*:]]
-; ATTRS-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T1]])
-; ATTRS-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T2]])
-; ATTRS-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T3]])
-; ATTRS-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T4]])
-; ATTRS-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T5]])
-; ATTRS-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T6]])
-; ATTRS-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T7]])
-; ATTRS-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T8]])
-; ATTRS-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T9]])
-; ATTRS-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T10]])
-; ATTRS-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T11]])
-; ATTRS-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T12]])
-; ATTRS-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T13]])
-; ATTRS-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T14]])
-; ATTRS-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T15]])
-; ATTRS-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T16]])
-; ATTRS-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T17]])
-; ATTRS-NEXT: [[U1:%.*]] = add i128 [[T1]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U1]])
-; ATTRS-NEXT: [[U2:%.*]] = add i128 [[T2]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U2]])
-; ATTRS-NEXT: [[U3:%.*]] = add i128 [[T3]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U3]])
-; ATTRS-NEXT: [[U4:%.*]] = add i128 [[T4]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U4]])
-; ATTRS-NEXT: [[U5:%.*]] = add i128 [[T5]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U5]])
-; ATTRS-NEXT: [[U6:%.*]] = add i128 [[T6]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U6]])
-; ATTRS-NEXT: [[U7:%.*]] = add i128 [[T7]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U7]])
-; ATTRS-NEXT: [[U8:%.*]] = add i128 [[T8]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U8]])
-; ATTRS-NEXT: [[U9:%.*]] = add i128 [[T9]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U9]])
-; ATTRS-NEXT: [[U10:%.*]] = add i128 [[T10]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U10]])
-; ATTRS-NEXT: [[U11:%.*]] = add i128 [[T11]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U11]])
-; ATTRS-NEXT: [[U12:%.*]] = add i128 [[T12]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U12]])
-; ATTRS-NEXT: [[U13:%.*]] = add i128 [[T13]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U13]])
-; ATTRS-NEXT: [[U14:%.*]] = add i128 [[T14]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U14]])
-; ATTRS-NEXT: [[U15:%.*]] = add i128 [[T15]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U15]])
-; ATTRS-NEXT: [[U16:%.*]] = add i128 [[T16]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U16]])
-; ATTRS-NEXT: [[U17:%.*]] = add i128 [[T17]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[U17]])
-; ATTRS-NEXT: call void @bar(i128 [[B1]])
-; ATTRS-NEXT: call void @bar(i128 [[B2]])
-; ATTRS-NEXT: call void @bar(i128 [[B3]])
-; ATTRS-NEXT: call void @bar(i128 [[B4]])
-; ATTRS-NEXT: call void @bar(i128 [[B5]])
-; ATTRS-NEXT: call void @bar(i128 [[B6]])
-; ATTRS-NEXT: call void @bar(i128 [[B7]])
-; ATTRS-NEXT: call void @bar(i128 [[B8]])
-; ATTRS-NEXT: call void @bar(i128 [[B9]])
-; ATTRS-NEXT: call void @bar(i128 [[B10]])
-; ATTRS-NEXT: call void @bar(i128 [[B11]])
-; ATTRS-NEXT: call void @bar(i128 [[B12]])
-; ATTRS-NEXT: call void @bar(i128 [[B13]])
-; ATTRS-NEXT: call void @bar(i128 [[B14]])
-; ATTRS-NEXT: call void @bar(i128 [[B15]])
-; ATTRS-NEXT: call void @bar(i128 [[B16]])
-; ATTRS-NEXT: call void @bar(i128 [[B17]])
-; ATTRS-NEXT: ret void
-;
-; NOFILTER-LABEL: define void @small_work_group(
-; NOFILTER-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR0:[0-9]+]] {
-; NOFILTER-NEXT: [[ENTRY:.*:]]
-; NOFILTER-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T1]])
-; NOFILTER-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T2]])
-; NOFILTER-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T3]])
-; NOFILTER-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T4]])
-; NOFILTER-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T5]])
-; NOFILTER-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T6]])
-; NOFILTER-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T7]])
-; NOFILTER-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T8]])
-; NOFILTER-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T9]])
-; NOFILTER-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T10]])
-; NOFILTER-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T11]])
-; NOFILTER-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T12]])
-; NOFILTER-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T13]])
-; NOFILTER-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T14]])
-; NOFILTER-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T15]])
-; NOFILTER-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T16]])
-; NOFILTER-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T17]])
-; NOFILTER-NEXT: [[U1:%.*]] = add i128 [[T1]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U1]])
-; NOFILTER-NEXT: [[U2:%.*]] = add i128 [[T2]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U2]])
-; NOFILTER-NEXT: [[U3:%.*]] = add i128 [[T3]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U3]])
-; NOFILTER-NEXT: [[U4:%.*]] = add i128 [[T4]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U4]])
-; NOFILTER-NEXT: [[U5:%.*]] = add i128 [[T5]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U5]])
-; NOFILTER-NEXT: [[U6:%.*]] = add i128 [[T6]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U6]])
-; NOFILTER-NEXT: [[U7:%.*]] = add i128 [[T7]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U7]])
-; NOFILTER-NEXT: [[U8:%.*]] = add i128 [[T8]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U8]])
-; NOFILTER-NEXT: [[U9:%.*]] = add i128 [[T9]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U9]])
-; NOFILTER-NEXT: [[U10:%.*]] = add i128 [[T10]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U10]])
-; NOFILTER-NEXT: [[U11:%.*]] = add i128 [[T11]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U11]])
-; NOFILTER-NEXT: [[U12:%.*]] = add i128 [[T12]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U12]])
-; NOFILTER-NEXT: [[U13:%.*]] = add i128 [[T13]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U13]])
-; NOFILTER-NEXT: [[U14:%.*]] = add i128 [[T14]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U14]])
-; NOFILTER-NEXT: [[U15:%.*]] = add i128 [[T15]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U15]])
-; NOFILTER-NEXT: [[U16:%.*]] = add i128 [[T16]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U16]])
-; NOFILTER-NEXT: [[U17:%.*]] = add i128 [[T17]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U17]])
-; NOFILTER-NEXT: call void @bar(i128 [[B1]])
-; NOFILTER-NEXT: call void @bar(i128 [[B2]])
-; NOFILTER-NEXT: call void @bar(i128 [[B3]])
-; NOFILTER-NEXT: call void @bar(i128 [[B4]])
-; NOFILTER-NEXT: call void @bar(i128 [[B5]])
-; NOFILTER-NEXT: call void @bar(i128 [[B6]])
-; NOFILTER-NEXT: call void @bar(i128 [[B7]])
-; NOFILTER-NEXT: call void @bar(i128 [[B8]])
-; NOFILTER-NEXT: call void @bar(i128 [[B9]])
-; NOFILTER-NEXT: call void @bar(i128 [[B10]])
-; NOFILTER-NEXT: call void @bar(i128 [[B11]])
-; NOFILTER-NEXT: call void @bar(i128 [[B12]])
-; NOFILTER-NEXT: call void @bar(i128 [[B13]])
-; NOFILTER-NEXT: call void @bar(i128 [[B14]])
-; NOFILTER-NEXT: call void @bar(i128 [[B15]])
-; NOFILTER-NEXT: call void @bar(i128 [[B16]])
-; NOFILTER-NEXT: call void @bar(i128 [[B17]])
-; NOFILTER-NEXT: ret void
-;
-entry:
- %s2 = shl i128 %s, 1
- %t1 = add i128 %b1, %s
- call void @bar(i128 %t1)
- %t2 = add i128 %b2, %s
- call void @bar(i128 %t2)
- %t3 = add i128 %b3, %s
- call void @bar(i128 %t3)
- %t4 = add i128 %b4, %s
- call void @bar(i128 %t4)
- %t5 = add i128 %b5, %s
- call void @bar(i128 %t5)
- %t6 = add i128 %b6, %s
- call void @bar(i128 %t6)
- %t7 = add i128 %b7, %s
- call void @bar(i128 %t7)
- %t8 = add i128 %b8, %s
- call void @bar(i128 %t8)
- %t9 = add i128 %b9, %s
- call void @bar(i128 %t9)
- %t10 = add i128 %b10, %s
- call void @bar(i128 %t10)
- %t11 = add i128 %b11, %s
- call void @bar(i128 %t11)
- %t12 = add i128 %b12, %s
- call void @bar(i128 %t12)
- %t13 = add i128 %b13, %s
- call void @bar(i128 %t13)
- %t14 = add i128 %b14, %s
- call void @bar(i128 %t14)
- %t15 = add i128 %b15, %s
- call void @bar(i128 %t15)
- %t16 = add i128 %b16, %s
- call void @bar(i128 %t16)
- %t17 = add i128 %b17, %s
- call void @bar(i128 %t17)
- %u1 = add i128 %b1, %s2
- call void @bar(i128 %u1)
- %u2 = add i128 %b2, %s2
- call void @bar(i128 %u2)
- %u3 = add i128 %b3, %s2
- call void @bar(i128 %u3)
- %u4 = add i128 %b4, %s2
- call void @bar(i128 %u4)
- %u5 = add i128 %b5, %s2
- call void @bar(i128 %u5)
- %u6 = add i128 %b6, %s2
- call void @bar(i128 %u6)
- %u7 = add i128 %b7, %s2
- call void @bar(i128 %u7)
- %u8 = add i128 %b8, %s2
- call void @bar(i128 %u8)
- %u9 = add i128 %b9, %s2
- call void @bar(i128 %u9)
- %u10 = add i128 %b10, %s2
- call void @bar(i128 %u10)
- %u11 = add i128 %b11, %s2
- call void @bar(i128 %u11)
- %u12 = add i128 %b12, %s2
- call void @bar(i128 %u12)
- %u13 = add i128 %b13, %s2
- call void @bar(i128 %u13)
- %u14 = add i128 %b14, %s2
- call void @bar(i128 %u14)
- %u15 = add i128 %b15, %s2
- call void @bar(i128 %u15)
- %u16 = add i128 %b16, %s2
- call void @bar(i128 %u16)
- %u17 = add i128 %b17, %s2
- call void @bar(i128 %u17)
- call void @bar(i128 %b1)
- call void @bar(i128 %b2)
- call void @bar(i128 %b3)
- call void @bar(i128 %b4)
- call void @bar(i128 %b5)
- call void @bar(i128 %b6)
- call void @bar(i128 %b7)
- call void @bar(i128 %b8)
- call void @bar(i128 %b9)
- call void @bar(i128 %b10)
- call void @bar(i128 %b11)
- call void @bar(i128 %b12)
- call void @bar(i128 %b13)
- call void @bar(i128 %b14)
- call void @bar(i128 %b15)
- call void @bar(i128 %b16)
- call void @bar(i128 %b17)
- ret void
-}
-
-; Same small work group, but asking for 8 waves per EU drops the budget to
-; 512/8 = 64, so 144 is over again and the rewrites are dropped. This is the
-; pair that shows waves-per-eu overriding what the work group size allows.
-define void @small_wg_high_occ(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) #1 {
-; BUDGET32-LABEL: define void @small_wg_high_occ(
-; BUDGET32-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR1:[0-9]+]] {
-; BUDGET32-NEXT: [[ENTRY:.*:]]
-; BUDGET32-NEXT: [[S2:%.*]] = shl i128 [[S]], 1
-; BUDGET32-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T1]])
-; BUDGET32-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T2]])
-; BUDGET32-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T3]])
-; BUDGET32-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T4]])
-; BUDGET32-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T5]])
-; BUDGET32-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T6]])
-; BUDGET32-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T7]])
-; BUDGET32-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T8]])
-; BUDGET32-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T9]])
-; BUDGET32-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T10]])
-; BUDGET32-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T11]])
-; BUDGET32-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T12]])
-; BUDGET32-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T13]])
-; BUDGET32-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T14]])
-; BUDGET32-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T15]])
-; BUDGET32-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T16]])
-; BUDGET32-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
-; BUDGET32-NEXT: call void @bar(i128 [[T17]])
-; BUDGET32-NEXT: [[U1:%.*]] = add i128 [[B1]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U1]])
-; BUDGET32-NEXT: [[U2:%.*]] = add i128 [[B2]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U2]])
-; BUDGET32-NEXT: [[U3:%.*]] = add i128 [[B3]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U3]])
-; BUDGET32-NEXT: [[U4:%.*]] = add i128 [[B4]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U4]])
-; BUDGET32-NEXT: [[U5:%.*]] = add i128 [[B5]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U5]])
-; BUDGET32-NEXT: [[U6:%.*]] = add i128 [[B6]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U6]])
-; BUDGET32-NEXT: [[U7:%.*]] = add i128 [[B7]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U7]])
-; BUDGET32-NEXT: [[U8:%.*]] = add i128 [[B8]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U8]])
-; BUDGET32-NEXT: [[U9:%.*]] = add i128 [[B9]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U9]])
-; BUDGET32-NEXT: [[U10:%.*]] = add i128 [[B10]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U10]])
-; BUDGET32-NEXT: [[U11:%.*]] = add i128 [[B11]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U11]])
-; BUDGET32-NEXT: [[U12:%.*]] = add i128 [[B12]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U12]])
-; BUDGET32-NEXT: [[U13:%.*]] = add i128 [[B13]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U13]])
-; BUDGET32-NEXT: [[U14:%.*]] = add i128 [[B14]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U14]])
-; BUDGET32-NEXT: [[U15:%.*]] = add i128 [[B15]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U15]])
-; BUDGET32-NEXT: [[U16:%.*]] = add i128 [[B16]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U16]])
-; BUDGET32-NEXT: [[U17:%.*]] = add i128 [[B17]], [[S2]]
-; BUDGET32-NEXT: call void @bar(i128 [[U17]])
-; BUDGET32-NEXT: call void @bar(i128 [[B1]])
-; BUDGET32-NEXT: call void @bar(i128 [[B2]])
-; BUDGET32-NEXT: call void @bar(i128 [[B3]])
-; BUDGET32-NEXT: call void @bar(i128 [[B4]])
-; BUDGET32-NEXT: call void @bar(i128 [[B5]])
-; BUDGET32-NEXT: call void @bar(i128 [[B6]])
-; BUDGET32-NEXT: call void @bar(i128 [[B7]])
-; BUDGET32-NEXT: call void @bar(i128 [[B8]])
-; BUDGET32-NEXT: call void @bar(i128 [[B9]])
-; BUDGET32-NEXT: call void @bar(i128 [[B10]])
-; BUDGET32-NEXT: call void @bar(i128 [[B11]])
-; BUDGET32-NEXT: call void @bar(i128 [[B12]])
-; BUDGET32-NEXT: call void @bar(i128 [[B13]])
-; BUDGET32-NEXT: call void @bar(i128 [[B14]])
-; BUDGET32-NEXT: call void @bar(i128 [[B15]])
-; BUDGET32-NEXT: call void @bar(i128 [[B16]])
-; BUDGET32-NEXT: call void @bar(i128 [[B17]])
-; BUDGET32-NEXT: ret void
-;
-; ATTRS-LABEL: define void @small_wg_high_occ(
-; ATTRS-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR1:[0-9]+]] {
-; ATTRS-NEXT: [[ENTRY:.*:]]
-; ATTRS-NEXT: [[S2:%.*]] = shl i128 [[S]], 1
-; ATTRS-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T1]])
-; ATTRS-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T2]])
-; ATTRS-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T3]])
-; ATTRS-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T4]])
-; ATTRS-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T5]])
-; ATTRS-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T6]])
-; ATTRS-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T7]])
-; ATTRS-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T8]])
-; ATTRS-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T9]])
-; ATTRS-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T10]])
-; ATTRS-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T11]])
-; ATTRS-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T12]])
-; ATTRS-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T13]])
-; ATTRS-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T14]])
-; ATTRS-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T15]])
-; ATTRS-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T16]])
-; ATTRS-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
-; ATTRS-NEXT: call void @bar(i128 [[T17]])
-; ATTRS-NEXT: [[U1:%.*]] = add i128 [[B1]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U1]])
-; ATTRS-NEXT: [[U2:%.*]] = add i128 [[B2]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U2]])
-; ATTRS-NEXT: [[U3:%.*]] = add i128 [[B3]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U3]])
-; ATTRS-NEXT: [[U4:%.*]] = add i128 [[B4]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U4]])
-; ATTRS-NEXT: [[U5:%.*]] = add i128 [[B5]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U5]])
-; ATTRS-NEXT: [[U6:%.*]] = add i128 [[B6]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U6]])
-; ATTRS-NEXT: [[U7:%.*]] = add i128 [[B7]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U7]])
-; ATTRS-NEXT: [[U8:%.*]] = add i128 [[B8]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U8]])
-; ATTRS-NEXT: [[U9:%.*]] = add i128 [[B9]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U9]])
-; ATTRS-NEXT: [[U10:%.*]] = add i128 [[B10]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U10]])
-; ATTRS-NEXT: [[U11:%.*]] = add i128 [[B11]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U11]])
-; ATTRS-NEXT: [[U12:%.*]] = add i128 [[B12]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U12]])
-; ATTRS-NEXT: [[U13:%.*]] = add i128 [[B13]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U13]])
-; ATTRS-NEXT: [[U14:%.*]] = add i128 [[B14]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U14]])
-; ATTRS-NEXT: [[U15:%.*]] = add i128 [[B15]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U15]])
-; ATTRS-NEXT: [[U16:%.*]] = add i128 [[B16]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U16]])
-; ATTRS-NEXT: [[U17:%.*]] = add i128 [[B17]], [[S2]]
-; ATTRS-NEXT: call void @bar(i128 [[U17]])
-; ATTRS-NEXT: call void @bar(i128 [[B1]])
-; ATTRS-NEXT: call void @bar(i128 [[B2]])
-; ATTRS-NEXT: call void @bar(i128 [[B3]])
-; ATTRS-NEXT: call void @bar(i128 [[B4]])
-; ATTRS-NEXT: call void @bar(i128 [[B5]])
-; ATTRS-NEXT: call void @bar(i128 [[B6]])
-; ATTRS-NEXT: call void @bar(i128 [[B7]])
-; ATTRS-NEXT: call void @bar(i128 [[B8]])
-; ATTRS-NEXT: call void @bar(i128 [[B9]])
-; ATTRS-NEXT: call void @bar(i128 [[B10]])
-; ATTRS-NEXT: call void @bar(i128 [[B11]])
-; ATTRS-NEXT: call void @bar(i128 [[B12]])
-; ATTRS-NEXT: call void @bar(i128 [[B13]])
-; ATTRS-NEXT: call void @bar(i128 [[B14]])
-; ATTRS-NEXT: call void @bar(i128 [[B15]])
-; ATTRS-NEXT: call void @bar(i128 [[B16]])
-; ATTRS-NEXT: call void @bar(i128 [[B17]])
-; ATTRS-NEXT: ret void
-;
-; NOFILTER-LABEL: define void @small_wg_high_occ(
-; NOFILTER-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR1:[0-9]+]] {
-; NOFILTER-NEXT: [[ENTRY:.*:]]
-; NOFILTER-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T1]])
-; NOFILTER-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T2]])
-; NOFILTER-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T3]])
-; NOFILTER-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T4]])
-; NOFILTER-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T5]])
-; NOFILTER-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T6]])
-; NOFILTER-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T7]])
-; NOFILTER-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T8]])
-; NOFILTER-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T9]])
-; NOFILTER-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T10]])
-; NOFILTER-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T11]])
-; NOFILTER-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T12]])
-; NOFILTER-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T13]])
-; NOFILTER-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T14]])
-; NOFILTER-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T15]])
-; NOFILTER-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T16]])
-; NOFILTER-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[T17]])
-; NOFILTER-NEXT: [[U1:%.*]] = add i128 [[T1]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U1]])
-; NOFILTER-NEXT: [[U2:%.*]] = add i128 [[T2]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U2]])
-; NOFILTER-NEXT: [[U3:%.*]] = add i128 [[T3]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U3]])
-; NOFILTER-NEXT: [[U4:%.*]] = add i128 [[T4]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U4]])
-; NOFILTER-NEXT: [[U5:%.*]] = add i128 [[T5]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U5]])
-; NOFILTER-NEXT: [[U6:%.*]] = add i128 [[T6]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U6]])
-; NOFILTER-NEXT: [[U7:%.*]] = add i128 [[T7]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U7]])
-; NOFILTER-NEXT: [[U8:%.*]] = add i128 [[T8]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U8]])
-; NOFILTER-NEXT: [[U9:%.*]] = add i128 [[T9]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U9]])
-; NOFILTER-NEXT: [[U10:%.*]] = add i128 [[T10]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U10]])
-; NOFILTER-NEXT: [[U11:%.*]] = add i128 [[T11]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U11]])
-; NOFILTER-NEXT: [[U12:%.*]] = add i128 [[T12]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U12]])
-; NOFILTER-NEXT: [[U13:%.*]] = add i128 [[T13]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U13]])
-; NOFILTER-NEXT: [[U14:%.*]] = add i128 [[T14]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U14]])
-; NOFILTER-NEXT: [[U15:%.*]] = add i128 [[T15]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U15]])
-; NOFILTER-NEXT: [[U16:%.*]] = add i128 [[T16]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U16]])
-; NOFILTER-NEXT: [[U17:%.*]] = add i128 [[T17]], [[S]]
-; NOFILTER-NEXT: call void @bar(i128 [[U17]])
-; NOFILTER-NEXT: call void @bar(i128 [[B1]])
-; NOFILTER-NEXT: call void @bar(i128 [[B2]])
-; NOFILTER-NEXT: call void @bar(i128 [[B3]])
-; NOFILTER-NEXT: call void @bar(i128 [[B4]])
-; NOFILTER-NEXT: call void @bar(i128 [[B5]])
-; NOFILTER-NEXT: call void @bar(i128 [[B6]])
-; NOFILTER-NEXT: call void @bar(i128 [[B7]])
-; NOFILTER-NEXT: call void @bar(i128 [[B8]])
-; NOFILTER-NEXT: call void @bar(i128 [[B9]])
-; NOFILTER-NEXT: call void @bar(i128 [[B10]])
-; NOFILTER-NEXT: call void @bar(i128 [[B11]])
-; NOFILTER-NEXT: call void @bar(i128 [[B12]])
-; NOFILTER-NEXT: call void @bar(i128 [[B13]])
-; NOFILTER-NEXT: call void @bar(i128 [[B14]])
-; NOFILTER-NEXT: call void @bar(i128 [[B15]])
-; NOFILTER-NEXT: call void @bar(i128 [[B16]])
-; NOFILTER-NEXT: call void @bar(i128 [[B17]])
-; NOFILTER-NEXT: ret void
+define void @peak_below_budget(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) #0 {
+; CHECK-LABEL: define void @peak_below_budget(
+; CHECK-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[T1:%.*]] = add i128 [[B1]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T1]])
+; CHECK-NEXT: [[T2:%.*]] = add i128 [[B2]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T2]])
+; CHECK-NEXT: [[T3:%.*]] = add i128 [[B3]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T3]])
+; CHECK-NEXT: [[T4:%.*]] = add i128 [[B4]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T4]])
+; CHECK-NEXT: [[T5:%.*]] = add i128 [[B5]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T5]])
+; CHECK-NEXT: [[T6:%.*]] = add i128 [[B6]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T6]])
+; CHECK-NEXT: [[T7:%.*]] = add i128 [[B7]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T7]])
+; CHECK-NEXT: [[T8:%.*]] = add i128 [[B8]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T8]])
+; CHECK-NEXT: [[T9:%.*]] = add i128 [[B9]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T9]])
+; CHECK-NEXT: [[T10:%.*]] = add i128 [[B10]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T10]])
+; CHECK-NEXT: [[T11:%.*]] = add i128 [[B11]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T11]])
+; CHECK-NEXT: [[T12:%.*]] = add i128 [[B12]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T12]])
+; CHECK-NEXT: [[T13:%.*]] = add i128 [[B13]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T13]])
+; CHECK-NEXT: [[T14:%.*]] = add i128 [[B14]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T14]])
+; CHECK-NEXT: [[T15:%.*]] = add i128 [[B15]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T15]])
+; CHECK-NEXT: [[T16:%.*]] = add i128 [[B16]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T16]])
+; CHECK-NEXT: [[T17:%.*]] = add i128 [[B17]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[T17]])
+; CHECK-NEXT: [[U1:%.*]] = add i128 [[T1]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U1]])
+; CHECK-NEXT: [[U2:%.*]] = add i128 [[T2]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U2]])
+; CHECK-NEXT: [[U3:%.*]] = add i128 [[T3]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U3]])
+; CHECK-NEXT: [[U4:%.*]] = add i128 [[T4]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U4]])
+; CHECK-NEXT: [[U5:%.*]] = add i128 [[T5]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U5]])
+; CHECK-NEXT: [[U6:%.*]] = add i128 [[T6]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U6]])
+; CHECK-NEXT: [[U7:%.*]] = add i128 [[T7]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U7]])
+; CHECK-NEXT: [[U8:%.*]] = add i128 [[T8]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U8]])
+; CHECK-NEXT: [[U9:%.*]] = add i128 [[T9]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U9]])
+; CHECK-NEXT: [[U10:%.*]] = add i128 [[T10]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U10]])
+; CHECK-NEXT: [[U11:%.*]] = add i128 [[T11]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U11]])
+; CHECK-NEXT: [[U12:%.*]] = add i128 [[T12]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U12]])
+; CHECK-NEXT: [[U13:%.*]] = add i128 [[T13]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U13]])
+; CHECK-NEXT: [[U14:%.*]] = add i128 [[T14]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U14]])
+; CHECK-NEXT: [[U15:%.*]] = add i128 [[T15]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U15]])
+; CHECK-NEXT: [[U16:%.*]] = add i128 [[T16]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U16]])
+; CHECK-NEXT: [[U17:%.*]] = add i128 [[T17]], [[S]]
+; CHECK-NEXT: call void @bar(i128 [[U17]])
+; CHECK-NEXT: call void @bar(i128 [[B1]])
+; CHECK-NEXT: call void @bar(i128 [[B2]])
+; CHECK-NEXT: call void @bar(i128 [[B3]])
+; CHECK-NEXT: call void @bar(i128 [[B4]])
+; CHECK-NEXT: call void @bar(i128 [[B5]])
+; CHECK-NEXT: call void @bar(i128 [[B6]])
+; CHECK-NEXT: call void @bar(i128 [[B7]])
+; CHECK-NEXT: call void @bar(i128 [[B8]])
+; CHECK-NEXT: call void @bar(i128 [[B9]])
+; CHECK-NEXT: call void @bar(i128 [[B10]])
+; CHECK-NEXT: call void @bar(i128 [[B11]])
+; CHECK-NEXT: call void @bar(i128 [[B12]])
+; CHECK-NEXT: call void @bar(i128 [[B13]])
+; CHECK-NEXT: call void @bar(i128 [[B14]])
+; CHECK-NEXT: call void @bar(i128 [[B15]])
+; CHECK-NEXT: call void @bar(i128 [[B16]])
+; CHECK-NEXT: call void @bar(i128 [[B17]])
+; CHECK-NEXT: ret void
;
entry:
%s2 = shl i128 %s, 1
@@ -1918,5 +391,4 @@ entry:
ret void
}
-attributes #0 = { "amdgpu-flat-work-group-size"="1,256" }
-attributes #1 = { "amdgpu-flat-work-group-size"="1,256" "amdgpu-waves-per-eu"="8" }
+attributes #0 = { "amdgpu-flat-work-group-size"="1,256" }
\ No newline at end of file
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
index c534ecb8a3fda..32761d2922acd 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
@@ -1,296 +1,114 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt < %s -passes=slsr -S | FileCheck %s --check-prefixes=CHECK,NOBUDGET
-; RUN: opt < %s -passes=slsr -S -slsr-reg-budget=32 | FileCheck %s --check-prefixes=CHECK,BUDGET
+; REQUIRES: asserts
+; RUN: opt -passes=slsr -stats -disable-output <%s 2>&1 | FileCheck %s
+
+; CHECK-NOT: Number of blocks whose rewrites SLSR skipped due to register pressure
; The register-pressure filter needs a register budget to compare against, and
; the generic TargetTransformInfo has none: getRegisterBudget() returns
; std::nullopt unless a target implements it. There is no target triple here, so
-; the filter returns early without even running its liveness or pressure
-; analyses, and every rewrite stands no matter how much pressure it adds.
+; RPFilter::run() returns early, before it even computes liveness or pressure,
+; and every rewrite stands.
;
-; @many_bases_overlapping is the same shape used in AMDGPU/slsr-rp-filter.ll:
-; 17 distinct bases whose rewrites take the block's peak pressure from 20 to 36
-; registers. Passing -slsr-reg-budget=32 supplies the missing budget, and with
-; the 0.9 safe fraction leaving 28 registers the rewrites are then dropped.
+; @many_basises_overlapping is the same body used by
+; @peak_above_budget in AMDGPU/slsr-rp-filter.ll: 17 distinct basises whose
+; rewrites take the block's peak pressure from 80 registers to 144. Under an
+; AMDGPU triple that exceeds the budget and the rewrites are dropped.
;
-; So the two runs differ only in whether a budget exists, which is what pins the
-; early return down: without one the filter is inert on targets that cannot
-; report a register count.
+; The check is on the statistic rather than on the IR because LLVM only prints a
+; statistic that was incremented at least once, so a missing counter means the
+; filter never skipped a block.
target datalayout = "e-i64:64-v16:16-v32:32-n16:32:64"
-declare void @foo(i32)
+declare void @bar(i128)
-define void @many_bases_overlapping(i32 %s, i32 %b1, i32 %b2, i32 %b3, i32 %b4, i32 %b5, i32 %b6, i32 %b7, i32 %b8, i32 %b9, i32 %b10, i32 %b11, i32 %b12, i32 %b13, i32 %b14, i32 %b15, i32 %b16, i32 %b17) {
-; NOBUDGET-LABEL: define void @many_bases_overlapping(
-; NOBUDGET-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; NOBUDGET-NEXT: [[ENTRY:.*:]]
-; NOBUDGET-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T1]])
-; NOBUDGET-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T2]])
-; NOBUDGET-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T3]])
-; NOBUDGET-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T4]])
-; NOBUDGET-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T5]])
-; NOBUDGET-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T6]])
-; NOBUDGET-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T7]])
-; NOBUDGET-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T8]])
-; NOBUDGET-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T9]])
-; NOBUDGET-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T10]])
-; NOBUDGET-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T11]])
-; NOBUDGET-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T12]])
-; NOBUDGET-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T13]])
-; NOBUDGET-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T14]])
-; NOBUDGET-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T15]])
-; NOBUDGET-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T16]])
-; NOBUDGET-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[T17]])
-; NOBUDGET-NEXT: [[U1:%.*]] = add i32 [[T1]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U1]])
-; NOBUDGET-NEXT: [[U2:%.*]] = add i32 [[T2]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U2]])
-; NOBUDGET-NEXT: [[U3:%.*]] = add i32 [[T3]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U3]])
-; NOBUDGET-NEXT: [[U4:%.*]] = add i32 [[T4]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U4]])
-; NOBUDGET-NEXT: [[U5:%.*]] = add i32 [[T5]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U5]])
-; NOBUDGET-NEXT: [[U6:%.*]] = add i32 [[T6]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U6]])
-; NOBUDGET-NEXT: [[U7:%.*]] = add i32 [[T7]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U7]])
-; NOBUDGET-NEXT: [[U8:%.*]] = add i32 [[T8]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U8]])
-; NOBUDGET-NEXT: [[U9:%.*]] = add i32 [[T9]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U9]])
-; NOBUDGET-NEXT: [[U10:%.*]] = add i32 [[T10]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U10]])
-; NOBUDGET-NEXT: [[U11:%.*]] = add i32 [[T11]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U11]])
-; NOBUDGET-NEXT: [[U12:%.*]] = add i32 [[T12]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U12]])
-; NOBUDGET-NEXT: [[U13:%.*]] = add i32 [[T13]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U13]])
-; NOBUDGET-NEXT: [[U14:%.*]] = add i32 [[T14]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U14]])
-; NOBUDGET-NEXT: [[U15:%.*]] = add i32 [[T15]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U15]])
-; NOBUDGET-NEXT: [[U16:%.*]] = add i32 [[T16]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U16]])
-; NOBUDGET-NEXT: [[U17:%.*]] = add i32 [[T17]], [[S]]
-; NOBUDGET-NEXT: call void @foo(i32 [[U17]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B1]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B2]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B3]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B4]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B5]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B6]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B7]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B8]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B9]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B10]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B11]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B12]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B13]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B14]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B15]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B16]])
-; NOBUDGET-NEXT: call void @foo(i32 [[B17]])
-; NOBUDGET-NEXT: ret void
-;
-; BUDGET-LABEL: define void @many_bases_overlapping(
-; BUDGET-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; BUDGET-NEXT: [[ENTRY:.*:]]
-; BUDGET-NEXT: [[S2:%.*]] = shl i32 [[S]], 1
-; BUDGET-NEXT: [[T1:%.*]] = add i32 [[B1]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T1]])
-; BUDGET-NEXT: [[T2:%.*]] = add i32 [[B2]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T2]])
-; BUDGET-NEXT: [[T3:%.*]] = add i32 [[B3]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T3]])
-; BUDGET-NEXT: [[T4:%.*]] = add i32 [[B4]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T4]])
-; BUDGET-NEXT: [[T5:%.*]] = add i32 [[B5]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T5]])
-; BUDGET-NEXT: [[T6:%.*]] = add i32 [[B6]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T6]])
-; BUDGET-NEXT: [[T7:%.*]] = add i32 [[B7]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T7]])
-; BUDGET-NEXT: [[T8:%.*]] = add i32 [[B8]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T8]])
-; BUDGET-NEXT: [[T9:%.*]] = add i32 [[B9]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T9]])
-; BUDGET-NEXT: [[T10:%.*]] = add i32 [[B10]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T10]])
-; BUDGET-NEXT: [[T11:%.*]] = add i32 [[B11]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T11]])
-; BUDGET-NEXT: [[T12:%.*]] = add i32 [[B12]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T12]])
-; BUDGET-NEXT: [[T13:%.*]] = add i32 [[B13]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T13]])
-; BUDGET-NEXT: [[T14:%.*]] = add i32 [[B14]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T14]])
-; BUDGET-NEXT: [[T15:%.*]] = add i32 [[B15]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T15]])
-; BUDGET-NEXT: [[T16:%.*]] = add i32 [[B16]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T16]])
-; BUDGET-NEXT: [[T17:%.*]] = add i32 [[B17]], [[S]]
-; BUDGET-NEXT: call void @foo(i32 [[T17]])
-; BUDGET-NEXT: [[U1:%.*]] = add i32 [[B1]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U1]])
-; BUDGET-NEXT: [[U2:%.*]] = add i32 [[B2]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U2]])
-; BUDGET-NEXT: [[U3:%.*]] = add i32 [[B3]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U3]])
-; BUDGET-NEXT: [[U4:%.*]] = add i32 [[B4]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U4]])
-; BUDGET-NEXT: [[U5:%.*]] = add i32 [[B5]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U5]])
-; BUDGET-NEXT: [[U6:%.*]] = add i32 [[B6]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U6]])
-; BUDGET-NEXT: [[U7:%.*]] = add i32 [[B7]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U7]])
-; BUDGET-NEXT: [[U8:%.*]] = add i32 [[B8]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U8]])
-; BUDGET-NEXT: [[U9:%.*]] = add i32 [[B9]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U9]])
-; BUDGET-NEXT: [[U10:%.*]] = add i32 [[B10]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U10]])
-; BUDGET-NEXT: [[U11:%.*]] = add i32 [[B11]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U11]])
-; BUDGET-NEXT: [[U12:%.*]] = add i32 [[B12]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U12]])
-; BUDGET-NEXT: [[U13:%.*]] = add i32 [[B13]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U13]])
-; BUDGET-NEXT: [[U14:%.*]] = add i32 [[B14]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U14]])
-; BUDGET-NEXT: [[U15:%.*]] = add i32 [[B15]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U15]])
-; BUDGET-NEXT: [[U16:%.*]] = add i32 [[B16]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U16]])
-; BUDGET-NEXT: [[U17:%.*]] = add i32 [[B17]], [[S2]]
-; BUDGET-NEXT: call void @foo(i32 [[U17]])
-; BUDGET-NEXT: call void @foo(i32 [[B1]])
-; BUDGET-NEXT: call void @foo(i32 [[B2]])
-; BUDGET-NEXT: call void @foo(i32 [[B3]])
-; BUDGET-NEXT: call void @foo(i32 [[B4]])
-; BUDGET-NEXT: call void @foo(i32 [[B5]])
-; BUDGET-NEXT: call void @foo(i32 [[B6]])
-; BUDGET-NEXT: call void @foo(i32 [[B7]])
-; BUDGET-NEXT: call void @foo(i32 [[B8]])
-; BUDGET-NEXT: call void @foo(i32 [[B9]])
-; BUDGET-NEXT: call void @foo(i32 [[B10]])
-; BUDGET-NEXT: call void @foo(i32 [[B11]])
-; BUDGET-NEXT: call void @foo(i32 [[B12]])
-; BUDGET-NEXT: call void @foo(i32 [[B13]])
-; BUDGET-NEXT: call void @foo(i32 [[B14]])
-; BUDGET-NEXT: call void @foo(i32 [[B15]])
-; BUDGET-NEXT: call void @foo(i32 [[B16]])
-; BUDGET-NEXT: call void @foo(i32 [[B17]])
-; BUDGET-NEXT: ret void
-;
+define void @many_basises_overlapping(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) {
entry:
- %s2 = shl i32 %s, 1
- %t1 = add i32 %b1, %s
- call void @foo(i32 %t1)
- %t2 = add i32 %b2, %s
- call void @foo(i32 %t2)
- %t3 = add i32 %b3, %s
- call void @foo(i32 %t3)
- %t4 = add i32 %b4, %s
- call void @foo(i32 %t4)
- %t5 = add i32 %b5, %s
- call void @foo(i32 %t5)
- %t6 = add i32 %b6, %s
- call void @foo(i32 %t6)
- %t7 = add i32 %b7, %s
- call void @foo(i32 %t7)
- %t8 = add i32 %b8, %s
- call void @foo(i32 %t8)
- %t9 = add i32 %b9, %s
- call void @foo(i32 %t9)
- %t10 = add i32 %b10, %s
- call void @foo(i32 %t10)
- %t11 = add i32 %b11, %s
- call void @foo(i32 %t11)
- %t12 = add i32 %b12, %s
- call void @foo(i32 %t12)
- %t13 = add i32 %b13, %s
- call void @foo(i32 %t13)
- %t14 = add i32 %b14, %s
- call void @foo(i32 %t14)
- %t15 = add i32 %b15, %s
- call void @foo(i32 %t15)
- %t16 = add i32 %b16, %s
- call void @foo(i32 %t16)
- %t17 = add i32 %b17, %s
- call void @foo(i32 %t17)
- %u1 = add i32 %b1, %s2
- call void @foo(i32 %u1)
- %u2 = add i32 %b2, %s2
- call void @foo(i32 %u2)
- %u3 = add i32 %b3, %s2
- call void @foo(i32 %u3)
- %u4 = add i32 %b4, %s2
- call void @foo(i32 %u4)
- %u5 = add i32 %b5, %s2
- call void @foo(i32 %u5)
- %u6 = add i32 %b6, %s2
- call void @foo(i32 %u6)
- %u7 = add i32 %b7, %s2
- call void @foo(i32 %u7)
- %u8 = add i32 %b8, %s2
- call void @foo(i32 %u8)
- %u9 = add i32 %b9, %s2
- call void @foo(i32 %u9)
- %u10 = add i32 %b10, %s2
- call void @foo(i32 %u10)
- %u11 = add i32 %b11, %s2
- call void @foo(i32 %u11)
- %u12 = add i32 %b12, %s2
- call void @foo(i32 %u12)
- %u13 = add i32 %b13, %s2
- call void @foo(i32 %u13)
- %u14 = add i32 %b14, %s2
- call void @foo(i32 %u14)
- %u15 = add i32 %b15, %s2
- call void @foo(i32 %u15)
- %u16 = add i32 %b16, %s2
- call void @foo(i32 %u16)
- %u17 = add i32 %b17, %s2
- call void @foo(i32 %u17)
- call void @foo(i32 %b1)
- call void @foo(i32 %b2)
- call void @foo(i32 %b3)
- call void @foo(i32 %b4)
- call void @foo(i32 %b5)
- call void @foo(i32 %b6)
- call void @foo(i32 %b7)
- call void @foo(i32 %b8)
- call void @foo(i32 %b9)
- call void @foo(i32 %b10)
- call void @foo(i32 %b11)
- call void @foo(i32 %b12)
- call void @foo(i32 %b13)
- call void @foo(i32 %b14)
- call void @foo(i32 %b15)
- call void @foo(i32 %b16)
- call void @foo(i32 %b17)
+ %s2 = shl i128 %s, 1
+ %t1 = add i128 %b1, %s
+ call void @bar(i128 %t1)
+ %t2 = add i128 %b2, %s
+ call void @bar(i128 %t2)
+ %t3 = add i128 %b3, %s
+ call void @bar(i128 %t3)
+ %t4 = add i128 %b4, %s
+ call void @bar(i128 %t4)
+ %t5 = add i128 %b5, %s
+ call void @bar(i128 %t5)
+ %t6 = add i128 %b6, %s
+ call void @bar(i128 %t6)
+ %t7 = add i128 %b7, %s
+ call void @bar(i128 %t7)
+ %t8 = add i128 %b8, %s
+ call void @bar(i128 %t8)
+ %t9 = add i128 %b9, %s
+ call void @bar(i128 %t9)
+ %t10 = add i128 %b10, %s
+ call void @bar(i128 %t10)
+ %t11 = add i128 %b11, %s
+ call void @bar(i128 %t11)
+ %t12 = add i128 %b12, %s
+ call void @bar(i128 %t12)
+ %t13 = add i128 %b13, %s
+ call void @bar(i128 %t13)
+ %t14 = add i128 %b14, %s
+ call void @bar(i128 %t14)
+ %t15 = add i128 %b15, %s
+ call void @bar(i128 %t15)
+ %t16 = add i128 %b16, %s
+ call void @bar(i128 %t16)
+ %t17 = add i128 %b17, %s
+ call void @bar(i128 %t17)
+ %u1 = add i128 %b1, %s2
+ call void @bar(i128 %u1)
+ %u2 = add i128 %b2, %s2
+ call void @bar(i128 %u2)
+ %u3 = add i128 %b3, %s2
+ call void @bar(i128 %u3)
+ %u4 = add i128 %b4, %s2
+ call void @bar(i128 %u4)
+ %u5 = add i128 %b5, %s2
+ call void @bar(i128 %u5)
+ %u6 = add i128 %b6, %s2
+ call void @bar(i128 %u6)
+ %u7 = add i128 %b7, %s2
+ call void @bar(i128 %u7)
+ %u8 = add i128 %b8, %s2
+ call void @bar(i128 %u8)
+ %u9 = add i128 %b9, %s2
+ call void @bar(i128 %u9)
+ %u10 = add i128 %b10, %s2
+ call void @bar(i128 %u10)
+ %u11 = add i128 %b11, %s2
+ call void @bar(i128 %u11)
+ %u12 = add i128 %b12, %s2
+ call void @bar(i128 %u12)
+ %u13 = add i128 %b13, %s2
+ call void @bar(i128 %u13)
+ %u14 = add i128 %b14, %s2
+ call void @bar(i128 %u14)
+ %u15 = add i128 %b15, %s2
+ call void @bar(i128 %u15)
+ %u16 = add i128 %b16, %s2
+ call void @bar(i128 %u16)
+ %u17 = add i128 %b17, %s2
+ call void @bar(i128 %u17)
+ call void @bar(i128 %b1)
+ call void @bar(i128 %b2)
+ call void @bar(i128 %b3)
+ call void @bar(i128 %b4)
+ call void @bar(i128 %b5)
+ call void @bar(i128 %b6)
+ call void @bar(i128 %b7)
+ call void @bar(i128 %b8)
+ call void @bar(i128 %b9)
+ call void @bar(i128 %b10)
+ call void @bar(i128 %b11)
+ call void @bar(i128 %b12)
+ call void @bar(i128 %b13)
+ call void @bar(i128 %b14)
+ call void @bar(i128 %b15)
+ call void @bar(i128 %b16)
+ call void @bar(i128 %b17)
ret void
}
-;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
-; CHECK: {{.*}}
>From 3538f0315b955cd2d39adfc147c0699045239d6e Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Thu, 3 Sep 2026 00:11:22 +0000
Subject: [PATCH 09/13] per-BB threshold on num basises
---
.../Scalar/StraightLineStrengthReduce.cpp | 131 +++++++++---------
...sr-rp-filter.ll => slsr-rewrite-filter.ll} | 12 +-
...sr-rp-filter.ll => slsr-rewrite-filter.ll} | 7 +-
3 files changed, 77 insertions(+), 73 deletions(-)
rename llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/{slsr-rp-filter.ll => slsr-rewrite-filter.ll} (97%)
rename llvm/test/Transforms/StraightLineStrengthReduce/{slsr-rp-filter.ll => slsr-rewrite-filter.ll} (92%)
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 45dc1942e1bb7..c5e803be9cc3c 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -111,8 +111,10 @@ using namespace llvm;
using namespace PatternMatch;
#define DEBUG_TYPE "slsr"
-#define DEBUG_SLSR_RP(X) DEBUG_WITH_TYPE(DEBUG_TYPE "-rp", X)
-#define DEBUG_SLSR_RP_DETAIL(X) DEBUG_WITH_TYPE(DEBUG_TYPE "-rp-detail", X)
+#define DEBUG_SLSR_REWRITE_FILTER(X) \
+ DEBUG_WITH_TYPE(DEBUG_TYPE "-rewrite-filter", X)
+#define DEBUG_SLSR_REWRITE_FILTER_DETAIL(X) \
+ DEBUG_WITH_TYPE(DEBUG_TYPE "-rewrite-filter-detail", X)
static const unsigned UnknownAddressSpace =
std::numeric_limits<unsigned>::max();
@@ -125,18 +127,15 @@ static cl::opt<bool>
EnablePoisonReuseGuard("enable-poison-reuse-guard", cl::init(true),
cl::desc("Enable poison-reuse guard"));
-static cl::opt<bool> EnableRPFilter(
- "slsr-rp-filter", cl::init(true), cl::Hidden,
+static cl::opt<bool> EnableRewriteFilter(
+ "slsr-rewrite-filter", cl::init(true), cl::Hidden,
cl::desc("SLSR: skip rewrites in blocks where they would push register "
"pressure past the target's register budget"));
STATISTIC(NumSCEVCandidateBasisDifferences,
"Number of candidate-basis SCEV differences computed by SLSR");
-STATISTIC(NumRPFilteredBlocks,
- "Number of blocks whose rewrites SLSR skipped due to register "
- "pressure");
-STATISTIC(NumRPFilteredCandidates,
- "Number of candidates SLSR skipped due to register pressure");
+STATISTIC(NumFilteredCandidates,
+ "Number of SLSR candidates not rewritten due to register pressure");
namespace {
@@ -1426,28 +1425,28 @@ bool StraightLineStrengthReduceLegacyPass::runOnFunction(Function &F) {
namespace {
-class RPFilter {
- // RPFilter targets one pathological shape: a block holding many distinct
- // bases. Each basis contributes one extended live range no matter how many
- // candidates are rewritten against it, so it is the number of distinct bases,
- // not the number of candidates, that tracks how many new concurrent live
- // ranges SLSR would create. Below this count no block in the function can
- // exhibit the pathology, and the liveness and pressure analyses are skipped.
- static constexpr unsigned MinDistinctBasesToFilter = 16;
+class RewriteFilter {
+ // Recondiser rewriting a basic block holding many distinct SLSR bases.
+
+ // Each basis contributes one extended live range no matter how many
+ // candidates are rewritten against it in a basic block.
+ // If the number of bases is above this threshold do a check if the
+ // liveness of the block has increase a lot by SLSR.
+ static constexpr unsigned MinDistinctBasisesToFilter = 16;
public:
using Candidate = StraightLineStrengthReduce::Candidate;
- RPFilter(const Function *F,
- DenseMap<Instruction *, Candidate *> &PickedCandidateMap,
- const TargetTransformInfo *TTI)
+ RewriteFilter(const Function *F,
+ DenseMap<Instruction *, Candidate *> &PickedCandidateMap,
+ const TargetTransformInfo *TTI)
: F(F), PickedCandidateMap(PickedCandidateMap), TTI(TTI) {}
// Return the candidates whose rewrite would push their block's register
// pressure past what the target can allocate.
DenseSet<const Instruction *> run() {
DenseSet<const Instruction *> InstsToSkip;
- if (!EnableRPFilter || PickedCandidateMap.empty())
+ if (!EnableRewriteFilter || PickedCandidateMap.empty())
return InstsToSkip;
// Without a budget there is nothing to compare the pressure against, so
@@ -1457,27 +1456,33 @@ class RPFilter {
return InstsToSkip;
buildBBToNumCandsAndBasises(PickedCandidateMap);
- if (MaxNumBasisesInBB <= MinDistinctBasesToFilter)
+ if (MaxNumBasisesInBB <= MinDistinctBasisesToFilter)
return InstsToSkip;
// Compute live-in and live-out of each BB in CFG
buildBBToLiveness(*F);
- DEBUG_SLSR_RP(dbgs() << "-- MaxRP of BBs -- \n");
+ DEBUG_SLSR_REWRITE_FILTER(dbgs() << "-- MaxRP of BBs -- \n");
SmallPtrSet<const BasicBlock *, 8> BBsToSkip;
for (auto &BB : *F) {
+
+ auto It = BBToNumCandsAndBasises.find(&BB);
+ if (It == BBToNumCandsAndBasises.end() ||
+ It->second.second <= MinDistinctBasisesToFilter)
+ continue;
+
const BlockLiveness &BL = getLiveness(&BB);
auto [MaxRP, MaxRPWithSLSR] =
- maxPressureInBlockBackward(BB, BL.LiveIn, BL.LiveOut);
- DEBUG_SLSR_RP(dbgs() << "MaxRP:" << BB.getName() << ": (" << MaxRP << ", "
- << MaxRPWithSLSR << ")" << "\n");
+ maxLivenessInBlockBackward(BB, BL.LiveIn, BL.LiveOut);
+ DEBUG_SLSR_REWRITE_FILTER(dbgs()
+ << "MaxRP:" << BB.getName() << ": (" << MaxRP
+ << ", " << MaxRPWithSLSR << ")" << "\n");
if (!rewriteWouldOverflowBudget(MaxRP, MaxRPWithSLSR, *Budget))
continue;
- DEBUG_SLSR_RP(dbgs() << "Skipping BB from SLSR: " << BB.getName()
- << "\n");
- ++NumRPFilteredBlocks;
+ DEBUG_SLSR_REWRITE_FILTER(
+ dbgs() << "Skipping BB from SLSR: " << BB.getName() << "\n");
BBsToSkip.insert(&BB);
} // Done with BBs
@@ -1487,7 +1492,7 @@ class RPFilter {
for (const auto &It : PickedCandidateMap)
if (BBsToSkip.contains(It.first->getParent()) &&
InstsToSkip.insert(It.first).second)
- ++NumRPFilteredCandidates;
+ NumFilteredCandidates++;
return InstsToSkip;
}
@@ -1508,6 +1513,9 @@ class RPFilter {
};
DenseMap<const BasicBlock *, BlockLiveness> BBToLiveness;
+ // The liveness scan asks for the weight of every live value at every
+ // instruction, so memoize on the type, which is all the weight depends on.
+ mutable DenseMap<Type *, unsigned> WeightCache;
// Liveness is only computed for functions that pass the candidate-count
// gate. Blocks of the remaining functions read as having nothing live across
@@ -1563,7 +1571,7 @@ class RPFilter {
}
}
MaxNumBasisesInBB = std::max(MaxNumBasisesInBB, UniqueBasises.size());
- DEBUG_WITH_TYPE("slsr-rp", {
+ DEBUG_SLSR_REWRITE_FILTER({
dbgs() << "BB: " << BB->getName() << " - NumCands: " << NumCands
<< " UniqueBasises: " << UniqueBasises.size() << "\n";
});
@@ -1584,29 +1592,25 @@ class RPFilter {
return isa<Instruction>(V) || isa<Argument>(V);
}
- // The pressure scan asks for the weight of every live value at every
- // instruction, so memoize on the type, which is all the weight depends on.
- mutable DenseMap<Type *, unsigned> RegWeightCache;
+ unsigned computeWeight(Type *Ty) const {
+ const DataLayout &DL = F->getDataLayout();
+
+ // TTI's getRegUsageForType is less accurate than
+ // default logic to compute RP for targets like AMDGPU
+ return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
+ }
- unsigned regWeight(const Value *V) const {
+ unsigned weight(const Value *V) const {
Type *Ty = V->getType();
if (Ty->isVoidTy() || Ty->isTokenTy())
return 0;
- auto [It, Inserted] = RegWeightCache.try_emplace(Ty);
+ auto [It, Inserted] = WeightCache.try_emplace(Ty);
if (Inserted)
- It->second = computeRegWeight(Ty);
+ It->second = computeWeight(Ty);
return It->second;
}
- unsigned computeRegWeight(Type *Ty) const {
- const DataLayout &DL = F->getDataLayout();
-
- // TTI's getRegUsageForType is less accurate than
- // default logic to compute RP for targets like AMDGPU
- return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
- }
-
void buildBBToLiveness(const Function &F) {
DenseMap<const BasicBlock *, ValueSet> UpExposed, Defs;
DenseMap<std::pair<const BasicBlock *, const BasicBlock *>, ValueSet>
@@ -1632,7 +1636,7 @@ class RPFilter {
if (!I.getType()->isVoidTy())
D.insert(&I);
}
- DEBUG_SLSR_RP_DETAIL({
+ DEBUG_SLSR_REWRITE_FILTER_DETAIL({
dbgs() << "BB-fill: " << BB.getName()
<< " UpExposed: " << UpExposed[&BB].size();
dbgs() << " Defs: " << Defs[&BB].size() << "\n";
@@ -1665,7 +1669,7 @@ class RPFilter {
Out.size() != BBToLiveness[BB].LiveOut.size())
Changed = true;
- DEBUG_SLSR_RP_DETAIL({
+ DEBUG_SLSR_REWRITE_FILTER_DETAIL({
dbgs() << "BB-update: " << BB->getName() << " In: " << In.size()
<< " Out: " << Out.size() << "\n";
});
@@ -1675,7 +1679,7 @@ class RPFilter {
}
}
- DEBUG_WITH_TYPE("slsr-rp", {
+ DEBUG_WITH_TYPE("slsr-rewrite-filter", {
dbgs() << "-- Live Ins/Outs of BBs -- \n";
for (const BasicBlock &BB : F) {
dbgs() << BB.getName() << ": ";
@@ -1712,15 +1716,13 @@ class RPFilter {
}
std::pair<unsigned, unsigned>
- maxPressureInBlockBackward(const BasicBlock &BB, const ValueSet &LiveIn,
+ maxLivenessInBlockBackward(const BasicBlock &BB, const ValueSet &LiveIn,
const ValueSet &LiveOut) const {
- // TODO: ValueSet is a SmallPtrSet, which is supposedly smaller than 33.
- // Could be a better-fitting data structure.
ValueSet LiveSet = LiveOut;
unsigned MaxW = 0;
for (const Value *V : LiveSet)
- MaxW += regWeight(V);
+ MaxW += weight(V);
// Initial LiveSetWithSLSR is the same as LiveOut.
// SLSR changes
@@ -1754,10 +1756,10 @@ class RPFilter {
// removed from LiveSetWithSLSR. As a heuristic, we don't dicern
// which Op of I is replaced by Basis. When IsCand is true, there is
// only one reg-like Op in practice.
- // TODO: Can further restrict by Candidate's type and DeltaKind.
LiveSetWithSLSR.erase(Op);
- DEBUG_SLSR_RP_DETAIL(dbgs() << "Removed Op from LiveSetWithSLSR: "
- << *Op << " in inst " << I << "\n");
+ DEBUG_SLSR_REWRITE_FILTER_DETAIL(
+ dbgs() << "Removed Op from LiveSetWithSLSR: " << *Op
+ << " in inst " << I << "\n");
} else {
LiveSetWithSLSR.insert(Op);
}
@@ -1770,13 +1772,13 @@ class RPFilter {
// Compute RP reaching for this instruction.
unsigned W = 0;
for (const Value *V : LiveSet) {
- auto RW = regWeight(V);
+ auto RW = weight(V);
W += RW;
}
MaxW = std::max(MaxW, W);
W = 0;
for (const Value *V : LiveSetWithSLSR) {
- auto RW = regWeight(V);
+ auto RW = weight(V);
W += RW;
}
MaxWWithSLSR = std::max(MaxWWithSLSR, W);
@@ -1810,6 +1812,7 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
}
sortCandidateInstructions();
+ // Keep picked candidates not to call pickRewriteCandidate() again.
DenseMap<Instruction *, Candidate *> PickedCandidateMap;
for (Instruction *I : SortedCandidateInsts)
if (Candidate *C = pickRewriteCandidate(I))
@@ -1818,18 +1821,18 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
// Candidates whose rewrite would push their block's register pressure past
// what the target can allocate. Evaluated on the original IR, before any
// rewriteCandidate mutates it: rewriting inserts instructions and calls
- // replaceAllUsesWith, which would invalidate the liveness and pressure
+ // replaceAllUsesWith, which would invalidate the liveness
// analyses the filter relies on.
- RPFilter RPFilter(&F, PickedCandidateMap, TTI);
- DenseSet<const Instruction *> ToSkipRewrite = RPFilter.run();
+ RewriteFilter RewriteFilter(&F, PickedCandidateMap, TTI);
+ DenseSet<const Instruction *> ToSkipRewrite = RewriteFilter.run();
// Rewrite candidates in the topological order that rewrites a Candidate
// always before rewriting its Basis
for (Instruction *I : reverse(SortedCandidateInsts)) {
- if (ToSkipRewrite.contains(I))
- continue;
- if (Candidate *C = pickRewriteCandidate(I))
- rewriteCandidate(*C);
+ auto It = PickedCandidateMap.find(I);
+ if (It != PickedCandidateMap.end())
+ if (!ToSkipRewrite.contains(It->first))
+ rewriteCandidate(*It->second);
}
for (auto *DeadIns : DeadInstructions)
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter.ll
similarity index 97%
rename from llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
rename to llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter.ll
index a5f86a1a42ce4..efcc3996a7ead 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter.ll
@@ -2,9 +2,9 @@
; RUN: opt < %s -mtriple=amdgpu-amdhsa-amd -passes=slsr -S | FileCheck %s
; SLSR rewrites a candidate as its basis plus a delta, which keeps the basis live
-; down to the candidate. The register-pressure filter drops a block's rewrites
-; when doing so would take the block's peak pressure past what the target can
-; allocate.
+; down to the candidate. (Multiple candidates can be rewritten off of one basis).
+; If the rewrites are likely to increase the basic block's peak pressure what
+; the target can allocate, the rewrites are skipped.
;
; The budget comes from TTI, which on AMDGPU reports the VGPR count implied by
; the occupancy the function is compiled for. This triple selects the generic
@@ -14,8 +14,8 @@
;
; Both functions below have byte-identical bodies: 17 distinct bases whose
; rewrites take the block's peak pressure from 80 registers to 144. The only
-; difference is the occupancy attribute, so the budget is the only variable and
-; the comparison isolates it.
+; difference is the occupancy attribute to demonstrate the difference with
+; different budgets
;
; @peak_above_budget no attribute. The work group size defaults to 1024
; threads, which is 16 wave64s over 4 EUs, so 4 waves per
@@ -391,4 +391,4 @@ entry:
ret void
}
-attributes #0 = { "amdgpu-flat-work-group-size"="1,256" }
\ No newline at end of file
+attributes #0 = { "amdgpu-flat-work-group-size"="1,256" }
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
similarity index 92%
rename from llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
rename to llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
index 32761d2922acd..fcc38a8dd43d5 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
@@ -1,16 +1,17 @@
+; NOTE: Do not auto-generate
; REQUIRES: asserts
; RUN: opt -passes=slsr -stats -disable-output <%s 2>&1 | FileCheck %s
-; CHECK-NOT: Number of blocks whose rewrites SLSR skipped due to register pressure
+; CHECK-NOT: Number of SLSR candidates not rewritten due to register pressure
-; The register-pressure filter needs a register budget to compare against, and
+; The SLSR rewirte filter needs a register budget to compare against, and
; the generic TargetTransformInfo has none: getRegisterBudget() returns
; std::nullopt unless a target implements it. There is no target triple here, so
; RPFilter::run() returns early, before it even computes liveness or pressure,
; and every rewrite stands.
;
; @many_basises_overlapping is the same body used by
-; @peak_above_budget in AMDGPU/slsr-rp-filter.ll: 17 distinct basises whose
+; @peak_above_budget in AMDGPU/slsr-rewrite-filter.ll: 17 distinct basises whose
; rewrites take the block's peak pressure from 80 registers to 144. Under an
; AMDGPU triple that exceeds the budget and the rewrites are dropped.
;
>From b0e285cea6bdf7cf9a65be169c1125e33537d9fa Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <Yoonseo.Choi at amd.com>
Date: Mon, 14 Sep 2026 18:36:07 -0500
Subject: [PATCH 10/13] Formatting
---
.../Scalar/StraightLineStrengthReduce.cpp | 25 ++++++++++---------
1 file changed, 13 insertions(+), 12 deletions(-)
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index c5e803be9cc3c..4ad67ac907f16 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -1462,7 +1462,7 @@ class RewriteFilter {
// Compute live-in and live-out of each BB in CFG
buildBBToLiveness(*F);
- DEBUG_SLSR_REWRITE_FILTER(dbgs() << "-- MaxRP of BBs -- \n");
+ DEBUG_SLSR_REWRITE_FILTER(dbgs() << "-- Max liveness of BBs -- \n");
SmallPtrSet<const BasicBlock *, 8> BBsToSkip;
for (auto &BB : *F) {
@@ -1472,13 +1472,14 @@ class RewriteFilter {
continue;
const BlockLiveness &BL = getLiveness(&BB);
- auto [MaxRP, MaxRPWithSLSR] =
+ auto [MaxLiveness, MaxLivenessWithSLSR] =
maxLivenessInBlockBackward(BB, BL.LiveIn, BL.LiveOut);
- DEBUG_SLSR_REWRITE_FILTER(dbgs()
- << "MaxRP:" << BB.getName() << ": (" << MaxRP
- << ", " << MaxRPWithSLSR << ")" << "\n");
+ DEBUG_SLSR_REWRITE_FILTER(dbgs() << "MaxLiveness:" << BB.getName()
+ << ": (" << MaxLiveness << ", "
+ << MaxLivenessWithSLSR << ")" << "\n");
- if (!rewriteWouldOverflowBudget(MaxRP, MaxRPWithSLSR, *Budget))
+ if (!rewriteWouldOverflowBudget(MaxLiveness, MaxLivenessWithSLSR,
+ *Budget))
continue;
DEBUG_SLSR_REWRITE_FILTER(
@@ -1540,8 +1541,8 @@ class RewriteFilter {
unsigned Budget) const {
// Leave the allocator some slack: it also has to satisfy register class
// and ABI constraints that this estimate knows nothing about.
- constexpr double SLSRRPSafeFraction = 0.9;
- unsigned SafeBudget = static_cast<unsigned>(Budget * SLSRRPSafeFraction);
+ constexpr double SafeRatio = 0.9;
+ unsigned SafeBudget = static_cast<unsigned>(Budget * SafeRatio);
// There is headroom, so however much the rewrite adds is irrelevant.
if (After <= SafeBudget)
@@ -1553,8 +1554,8 @@ class RewriteFilter {
// Already over budget. SLSR can still lower pressure here, so only refuse
// rewrites that make it meaningfully worse.
- constexpr unsigned SLSRRPAbsDelta = 4;
- return After > Before && After - Before > SLSRRPAbsDelta;
+ constexpr unsigned AbsDelta = 4;
+ return After > Before && After - Before > AbsDelta;
}
std::pair<unsigned, unsigned> countCandsAndBasisesInBB(
@@ -1596,7 +1597,7 @@ class RewriteFilter {
const DataLayout &DL = F->getDataLayout();
// TTI's getRegUsageForType is less accurate than
- // default logic to compute RP for targets like AMDGPU
+ // default logic to compute pressure for targets like AMDGPU
return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
}
@@ -1769,7 +1770,7 @@ class RewriteFilter {
SeenLastUse.insert(Op);
}
- // Compute RP reaching for this instruction.
+ // Compute weight reaching for this instruction.
unsigned W = 0;
for (const Value *V : LiveSet) {
auto RW = weight(V);
>From aea1cc9687fc641b01b987db58d6e54b156b7bf6 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <Yoonseo.Choi at amd.com>
Date: Wed, 23 Sep 2026 21:14:05 -0500
Subject: [PATCH 11/13] Update logic for a safe budget. Fixed some typos.
---
.../Scalar/StraightLineStrengthReduce.cpp | 15 +++++++--------
.../slsr-rewrite-filter.ll | 4 ++--
2 files changed, 9 insertions(+), 10 deletions(-)
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 4ad67ac907f16..20b5376fca492 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -136,6 +136,7 @@ STATISTIC(NumSCEVCandidateBasisDifferences,
"Number of candidate-basis SCEV differences computed by SLSR");
STATISTIC(NumFilteredCandidates,
"Number of SLSR candidates not rewritten due to register pressure");
+STATISTIC(NumRewrittenCandidates, "Number of SLSR candidates rewritten");
namespace {
@@ -1426,7 +1427,7 @@ bool StraightLineStrengthReduceLegacyPass::runOnFunction(Function &F) {
namespace {
class RewriteFilter {
- // Recondiser rewriting a basic block holding many distinct SLSR bases.
+ // Reconsider rewriting a basic block holding many distinct SLSR bases.
// Each basis contributes one extended live range no matter how many
// candidates are rewritten against it in a basic block.
@@ -1465,7 +1466,6 @@ class RewriteFilter {
DEBUG_SLSR_REWRITE_FILTER(dbgs() << "-- Max liveness of BBs -- \n");
SmallPtrSet<const BasicBlock *, 8> BBsToSkip;
for (auto &BB : *F) {
-
auto It = BBToNumCandsAndBasises.find(&BB);
if (It == BBToNumCandsAndBasises.end() ||
It->second.second <= MinDistinctBasisesToFilter)
@@ -1518,9 +1518,6 @@ class RewriteFilter {
// instruction, so memoize on the type, which is all the weight depends on.
mutable DenseMap<Type *, unsigned> WeightCache;
- // Liveness is only computed for functions that pass the candidate-count
- // gate. Blocks of the remaining functions read as having nothing live across
- // their boundaries, which is why no rewrite is suppressed in that case.
const BlockLiveness &getLiveness(const BasicBlock *BB) const {
static const BlockLiveness Empty;
auto It = BBToLiveness.find(BB);
@@ -1541,8 +1538,8 @@ class RewriteFilter {
unsigned Budget) const {
// Leave the allocator some slack: it also has to satisfy register class
// and ABI constraints that this estimate knows nothing about.
- constexpr double SafeRatio = 0.9;
- unsigned SafeBudget = static_cast<unsigned>(Budget * SafeRatio);
+ constexpr unsigned SafeMargin = 8;
+ unsigned SafeBudget = Budget >= SafeMargin ? Budget - SafeMargin : Budget;
// There is headroom, so however much the rewrite adds is irrelevant.
if (After <= SafeBudget)
@@ -1832,8 +1829,10 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
for (Instruction *I : reverse(SortedCandidateInsts)) {
auto It = PickedCandidateMap.find(I);
if (It != PickedCandidateMap.end())
- if (!ToSkipRewrite.contains(It->first))
+ if (!ToSkipRewrite.contains(It->first)) {
rewriteCandidate(*It->second);
+ NumRewrittenCandidates++;
+ }
}
for (auto *DeadIns : DeadInstructions)
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
index fcc38a8dd43d5..66379fd2b01c0 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
@@ -4,10 +4,10 @@
; CHECK-NOT: Number of SLSR candidates not rewritten due to register pressure
-; The SLSR rewirte filter needs a register budget to compare against, and
+; The SLSR rewrite filter needs a register budget to compare against, and
; the generic TargetTransformInfo has none: getRegisterBudget() returns
; std::nullopt unless a target implements it. There is no target triple here, so
-; RPFilter::run() returns early, before it even computes liveness or pressure,
+; RewriteFilter::run() returns early, before it even computes liveness or pressure,
; and every rewrite stands.
;
; @many_basises_overlapping is the same body used by
>From e921b67607b2ed38bc001ac35c5aed5a1e38472a Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <Yoonseo.Choi at amd.com>
Date: Fri, 25 Sep 2026 11:26:08 -0500
Subject: [PATCH 12/13] Bail out on Scalable type; Remove CHECK-NOT
---
.../Scalar/StraightLineStrengthReduce.cpp | 69 ++++++++----
.../AMDGPU/slsr-rewrite-filter-scalable.ll | 103 ++++++++++++++++++
.../AMDGPU/slsr-rewrite-filter.ll | 9 ++
.../slsr-rewrite-filter.ll | 13 +--
4 files changed, 164 insertions(+), 30 deletions(-)
create mode 100644 llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter-scalable.ll
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 20b5376fca492..d09b5c8da9e88 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -1472,8 +1472,13 @@ class RewriteFilter {
continue;
const BlockLiveness &BL = getLiveness(&BB);
- auto [MaxLiveness, MaxLivenessWithSLSR] =
- maxLivenessInBlockBackward(BB, BL.LiveIn, BL.LiveOut);
+ auto Liveness = maxLivenessInBlockBackward(BB, BL.LiveIn, BL.LiveOut);
+ // A block holding a value with no fixed register footprint cannot be
+ // weighed against the budget, so leave its rewrites alone.
+ if (!Liveness)
+ continue;
+
+ auto [MaxLiveness, MaxLivenessWithSLSR] = *Liveness;
DEBUG_SLSR_REWRITE_FILTER(dbgs() << "MaxLiveness:" << BB.getName()
<< ": (" << MaxLiveness << ", "
<< MaxLivenessWithSLSR << ")" << "\n");
@@ -1516,7 +1521,7 @@ class RewriteFilter {
DenseMap<const BasicBlock *, BlockLiveness> BBToLiveness;
// The liveness scan asks for the weight of every live value at every
// instruction, so memoize on the type, which is all the weight depends on.
- mutable DenseMap<Type *, unsigned> WeightCache;
+ mutable DenseMap<Type *, std::optional<unsigned>> WeightCache;
const BlockLiveness &getLiveness(const BasicBlock *BB) const {
static const BlockLiveness Empty;
@@ -1590,15 +1595,23 @@ class RewriteFilter {
return isa<Instruction>(V) || isa<Argument>(V);
}
- unsigned computeWeight(Type *Ty) const {
+ // Returns std::nullopt for a type with no fixed register footprint, which is
+ // the signal to stop weighing the block.
+ std::optional<unsigned> computeWeight(Type *Ty) const {
const DataLayout &DL = F->getDataLayout();
+ // A scalable type occupies vscale times its minimum size, which is a
+ // runtime quantity, so it cannot be compared against a fixed budget.
+ TypeSize Size = DL.getTypeSizeInBits(Ty);
+ if (Size.isScalable())
+ return std::nullopt;
+
// TTI's getRegUsageForType is less accurate than
// default logic to compute pressure for targets like AMDGPU
- return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
+ return divideCeil(Size.getFixedValue(), 32);
}
- unsigned weight(const Value *V) const {
+ std::optional<unsigned> weight(const Value *V) const {
Type *Ty = V->getType();
if (Ty->isVoidTy() || Ty->isTokenTy())
return 0;
@@ -1713,14 +1726,27 @@ class RewriteFilter {
return false;
}
- std::pair<unsigned, unsigned>
+ // Returns std::nullopt when a live value has no fixed register footprint, so
+ // the block's pressure cannot be compared against the budget.
+ std::optional<std::pair<unsigned, unsigned>>
maxLivenessInBlockBackward(const BasicBlock &BB, const ValueSet &LiveIn,
const ValueSet &LiveOut) const {
+ auto SumWeights = [this](const ValueSet &Live) -> std::optional<unsigned> {
+ unsigned W = 0;
+ for (const Value *V : Live) {
+ std::optional<unsigned> RW = weight(V);
+ if (!RW)
+ return std::nullopt;
+ W += *RW;
+ }
+ return W;
+ };
ValueSet LiveSet = LiveOut;
- unsigned MaxW = 0;
- for (const Value *V : LiveSet)
- MaxW += weight(V);
+ std::optional<unsigned> InitialW = SumWeights(LiveSet);
+ if (!InitialW)
+ return std::nullopt;
+ unsigned MaxW = *InitialW;
// Initial LiveSetWithSLSR is the same as LiveOut.
// SLSR changes
@@ -1768,18 +1794,15 @@ class RewriteFilter {
}
// Compute weight reaching for this instruction.
- unsigned W = 0;
- for (const Value *V : LiveSet) {
- auto RW = weight(V);
- W += RW;
- }
- MaxW = std::max(MaxW, W);
- W = 0;
- for (const Value *V : LiveSetWithSLSR) {
- auto RW = weight(V);
- W += RW;
- }
- MaxWWithSLSR = std::max(MaxWWithSLSR, W);
+ std::optional<unsigned> W = SumWeights(LiveSet);
+ if (!W)
+ return std::nullopt;
+ std::optional<unsigned> WWithSLSR = SumWeights(LiveSetWithSLSR);
+ if (!WWithSLSR)
+ return std::nullopt;
+
+ MaxW = std::max(MaxW, *W);
+ MaxWWithSLSR = std::max(MaxWWithSLSR, *WWithSLSR);
// Remove the def (Instruction) from LiveSet for the next upward
// instruction.
@@ -1789,7 +1812,7 @@ class RewriteFilter {
}
}
- return {MaxW, MaxWWithSLSR};
+ return {{MaxW, MaxWWithSLSR}};
}
};
} // end of anonymous namespace
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter-scalable.ll b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter-scalable.ll
new file mode 100644
index 0000000000000..0c4be16d37232
--- /dev/null
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter-scalable.ll
@@ -0,0 +1,103 @@
+; NOTE: Do not auto-generate
+; REQUIRES: asserts
+; RUN: opt < %s -mtriple=amdgpu-amdhsa-amd -passes=slsr -stats -disable-output 2>&1 | FileCheck %s
+
+; The body is @peak_above_budget from slsr-rewrite-filter.ll, which is filtered
+; under this triple, plus one scalable value live across the block. All 17
+; candidates being rewritten. SLSR's RewriteFilter bails out on scalable types.
+; CHECK: {{^ *}}17 slsr{{ +}}- Number of SLSR candidates rewritten
+
+declare void @bar(i128)
+declare void @vec_use(<vscale x 4 x i32>)
+
+define void @scalable_live_value(<vscale x 4 x i32> %v, i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) {
+entry:
+ %s2 = shl i128 %s, 1
+ %t1 = add i128 %b1, %s
+ call void @bar(i128 %t1)
+ %t2 = add i128 %b2, %s
+ call void @bar(i128 %t2)
+ %t3 = add i128 %b3, %s
+ call void @bar(i128 %t3)
+ %t4 = add i128 %b4, %s
+ call void @bar(i128 %t4)
+ %t5 = add i128 %b5, %s
+ call void @bar(i128 %t5)
+ %t6 = add i128 %b6, %s
+ call void @bar(i128 %t6)
+ %t7 = add i128 %b7, %s
+ call void @bar(i128 %t7)
+ %t8 = add i128 %b8, %s
+ call void @bar(i128 %t8)
+ %t9 = add i128 %b9, %s
+ call void @bar(i128 %t9)
+ %t10 = add i128 %b10, %s
+ call void @bar(i128 %t10)
+ %t11 = add i128 %b11, %s
+ call void @bar(i128 %t11)
+ %t12 = add i128 %b12, %s
+ call void @bar(i128 %t12)
+ %t13 = add i128 %b13, %s
+ call void @bar(i128 %t13)
+ %t14 = add i128 %b14, %s
+ call void @bar(i128 %t14)
+ %t15 = add i128 %b15, %s
+ call void @bar(i128 %t15)
+ %t16 = add i128 %b16, %s
+ call void @bar(i128 %t16)
+ %t17 = add i128 %b17, %s
+ call void @bar(i128 %t17)
+ %u1 = add i128 %b1, %s2
+ call void @bar(i128 %u1)
+ %u2 = add i128 %b2, %s2
+ call void @bar(i128 %u2)
+ %u3 = add i128 %b3, %s2
+ call void @bar(i128 %u3)
+ %u4 = add i128 %b4, %s2
+ call void @bar(i128 %u4)
+ %u5 = add i128 %b5, %s2
+ call void @bar(i128 %u5)
+ %u6 = add i128 %b6, %s2
+ call void @bar(i128 %u6)
+ %u7 = add i128 %b7, %s2
+ call void @bar(i128 %u7)
+ %u8 = add i128 %b8, %s2
+ call void @bar(i128 %u8)
+ %u9 = add i128 %b9, %s2
+ call void @bar(i128 %u9)
+ %u10 = add i128 %b10, %s2
+ call void @bar(i128 %u10)
+ %u11 = add i128 %b11, %s2
+ call void @bar(i128 %u11)
+ %u12 = add i128 %b12, %s2
+ call void @bar(i128 %u12)
+ %u13 = add i128 %b13, %s2
+ call void @bar(i128 %u13)
+ %u14 = add i128 %b14, %s2
+ call void @bar(i128 %u14)
+ %u15 = add i128 %b15, %s2
+ call void @bar(i128 %u15)
+ %u16 = add i128 %b16, %s2
+ call void @bar(i128 %u16)
+ %u17 = add i128 %b17, %s2
+ call void @bar(i128 %u17)
+ call void @bar(i128 %b1)
+ call void @bar(i128 %b2)
+ call void @bar(i128 %b3)
+ call void @bar(i128 %b4)
+ call void @bar(i128 %b5)
+ call void @bar(i128 %b6)
+ call void @bar(i128 %b7)
+ call void @bar(i128 %b8)
+ call void @bar(i128 %b9)
+ call void @bar(i128 %b10)
+ call void @bar(i128 %b11)
+ call void @bar(i128 %b12)
+ call void @bar(i128 %b13)
+ call void @bar(i128 %b14)
+ call void @bar(i128 %b15)
+ call void @bar(i128 %b16)
+ call void @bar(i128 %b17)
+ call void @vec_use(<vscale x 4 x i32> %v)
+ ret void
+}
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter.ll
index efcc3996a7ead..49e80a3b64615 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter.ll
@@ -1,5 +1,14 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
; RUN: opt < %s -mtriple=amdgpu-amdhsa-amd -passes=slsr -S | FileCheck %s
+; REQUIRES: asserts
+; RUN: opt < %s -mtriple=amdgpu-amdhsa-amd -passes=slsr -stats -disable-output 2>&1 \
+; RUN: | FileCheck %s --check-prefix=STATS
+
+; Statistics are module wide, so the two counters below separate the two
+; functions: the 17 skipped candidates come from @peak_above_budget and the 17
+; rewritten ones from @peak_below_budget.
+; STATS: {{^ *}}17 slsr{{ +}}- Number of SLSR candidates not rewritten due to register pressure
+; STATS: {{^ *}}17 slsr{{ +}}- Number of SLSR candidates rewritten
; SLSR rewrites a candidate as its basis plus a delta, which keeps the basis live
; down to the candidate. (Multiple candidates can be rewritten off of one basis).
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
index 66379fd2b01c0..d1609bc392326 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
@@ -2,7 +2,7 @@
; REQUIRES: asserts
; RUN: opt -passes=slsr -stats -disable-output <%s 2>&1 | FileCheck %s
-; CHECK-NOT: Number of SLSR candidates not rewritten due to register pressure
+; CHECK: {{^ *}}17 slsr{{ +}}- Number of SLSR candidates rewritten
; The SLSR rewrite filter needs a register budget to compare against, and
; the generic TargetTransformInfo has none: getRegisterBudget() returns
@@ -10,20 +10,19 @@
; RewriteFilter::run() returns early, before it even computes liveness or pressure,
; and every rewrite stands.
;
-; @many_basises_overlapping is the same body used by
-; @peak_above_budget in AMDGPU/slsr-rewrite-filter.ll: 17 distinct basises whose
+; @many_bases_overlapping is the same body used by
+; @peak_above_budget in AMDGPU/slsr-rewrite-filter.ll: 17 distinct bases whose
; rewrites take the block's peak pressure from 80 registers to 144. Under an
; AMDGPU triple that exceeds the budget and the rewrites are dropped.
;
-; The check is on the statistic rather than on the IR because LLVM only prints a
-; statistic that was incremented at least once, so a missing counter means the
-; filter never skipped a block.
+; All 17 candidates being rewritten is what shows the filter never skipped a
+; block.
target datalayout = "e-i64:64-v16:16-v32:32-n16:32:64"
declare void @bar(i128)
-define void @many_basises_overlapping(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) {
+define void @many_bases_overlapping(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) {
entry:
%s2 = shl i128 %s, 1
%t1 = add i128 %b1, %s
>From 595f4be5526e9efa9746715121a15af1c9d2e48a Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <Yoonseo.Choi at amd.com>
Date: Thu, 1 Oct 2026 22:22:09 -0500
Subject: [PATCH 13/13] Move AMDGPU specific logics into TTI interface; clarify
comments
---
.../llvm/Analysis/TargetTransformInfo.h | 19 +++++++++++--------
.../AMDGPU/AMDGPUTargetTransformInfo.cpp | 6 +++++-
.../Scalar/StraightLineStrengthReduce.cpp | 18 +++++++-----------
3 files changed, 23 insertions(+), 20 deletions(-)
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index e1d6c692f1c3b..413b0f126acac 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -1359,15 +1359,18 @@ class TargetTransformInfo {
/// \return the number of registers in the target-provided register class.
LLVM_ABI unsigned getNumberOfRegisters(unsigned ClassID) const;
- /// \return The number of registers available to \p F before the register
- /// allocator is forced to spill, or std::nullopt if the target cannot
- /// provide a meaningful bound.
+ /// \return A conservative bound on the number of registers \p F can use
+ /// before the register allocator is likely to spill, or std::nullopt if the
+ /// target has no meaningful bound to report.
///
- /// Unlike getNumberOfRegisters(), which some targets deliberately
- /// under-report to tune vectorization and interleaving, this is meant to be
- /// a real budget usable by register-pressure heuristics. Targets with
- /// several register files report the budget for the file that dominates
- /// pressure.
+ /// The budget is a property of the function, not only of the subtarget: it
+ /// may depend on attributes that constrain how many registers the function
+ /// is permitted to use. So two functions in the same module can have
+ /// different budgets.
+ ///
+ /// It is intended for heuristics deciding whether a
+ /// transform is about to make register pressure a problem, and staying below.
+ /// Not a guarantee that no spilling occurs.
LLVM_ABI std::optional<unsigned> getRegisterBudget(const Function &F) const;
/// \return true if the target supports load/store that enables fault
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 16555675edda3..8c0a0897508ef 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -313,7 +313,11 @@ std::optional<unsigned> GCNTTIImpl::getRegisterBudget(const Function &F) const {
// Report the VGPR budget implied by the occupancy F is compiled for. Callers
// comparing a single lumped pressure number against this should be
// conservative on the SGPR side, which is intentional.
- return ST->getMaxNumVGPRs(F);
+ // Leave the allocator some slack.
+ constexpr unsigned SafeMargin = 8;
+ unsigned Budget = ST->getMaxNumVGPRs(F);
+ unsigned SafeBudget = Budget >= SafeMargin ? Budget - SafeMargin : Budget;
+ return SafeBudget;
}
TypeSize
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index d09b5c8da9e88..3fce004e75b97 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -1541,23 +1541,16 @@ class RewriteFilter {
// registers than \p Budget.
bool rewriteWouldOverflowBudget(unsigned Before, unsigned After,
unsigned Budget) const {
- // Leave the allocator some slack: it also has to satisfy register class
- // and ABI constraints that this estimate knows nothing about.
- constexpr unsigned SafeMargin = 8;
- unsigned SafeBudget = Budget >= SafeMargin ? Budget - SafeMargin : Budget;
// There is headroom, so however much the rewrite adds is irrelevant.
- if (After <= SafeBudget)
+ if (After <= Budget)
return false;
// The rewrite is what takes the block over.
- if (Before <= SafeBudget)
+ if (Before <= Budget)
return true;
- // Already over budget. SLSR can still lower pressure here, so only refuse
- // rewrites that make it meaningfully worse.
- constexpr unsigned AbsDelta = 4;
- return After > Before && After - Before > AbsDelta;
+ return After > Before && After - Before > 0;
}
std::pair<unsigned, unsigned> countCandsAndBasisesInBB(
@@ -1608,7 +1601,10 @@ class RewriteFilter {
// TTI's getRegUsageForType is less accurate than
// default logic to compute pressure for targets like AMDGPU
- return divideCeil(Size.getFixedValue(), 32);
+ unsigned RegisterBitWidth =
+ TTI->getRegisterBitWidth(TargetTransformInfo::RGK_Scalar)
+ .getFixedValue();
+ return divideCeil(Size.getFixedValue(), RegisterBitWidth);
}
std::optional<unsigned> weight(const Value *V) const {
More information about the llvm-commits
mailing list