[llvm] [SLSR] Skipping rewriting based on liveness (PR #218470)

Yoonseo Choi via llvm-commits llvm-commits at lists.llvm.org
Thu Oct 1 21:25:38 PDT 2026


https://github.com/yoonseoch updated https://github.com/llvm/llvm-project/pull/218470

>From 983f3c757a8df3bbd213983b866f1f86af98ac3e Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Tue, 4 Aug 2026 00:07:56 +0000
Subject: [PATCH 01/13] [SLSR] Adding a cost model to avoid high regpressure

---
 .../Scalar/StraightLineStrengthReduce.cpp     | 169 +++++++++++++++++-
 .../slsr-basis-distance-threshold.ll          |  51 ++++++
 2 files changed, 219 insertions(+), 1 deletion(-)
 create mode 100644 llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll

diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 4fa462da9ec47..31fe051cca640 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -122,8 +122,15 @@ static cl::opt<bool>
     EnablePoisonReuseGuard("enable-poison-reuse-guard", cl::init(true),
                            cl::desc("Enable poison-reuse guard"));
 
+<<<<<<< HEAD
 STATISTIC(NumSCEVCandidateBasisDifferences,
           "Number of candidate-basis SCEV differences computed by SLSR");
+=======
+static cl::opt<int> SLSRBasisDistanceThreshold(
+    "slsr-basis-distance-threshold", cl::init(96), cl::Hidden,
+    cl::desc("SLSR: skip rewrite if in-block distance from Basis's last "
+             "same-block use to the candidate Inst exceeds this"));
+>>>>>>> a5630b19d18c ([SLSR] Adding a cost model to avoid high regpressure)
 
 namespace {
 
@@ -592,6 +599,15 @@ class StraightLineStrengthReduce {
     if (auto *StrideInst = dyn_cast<Instruction>(C.Stride))
       PropagateDependency(StrideInst);
   };
+
+  bool hasOperandsUsedInNonRewritableUsersInAnotherBlock(
+      llvm::Instruction *Inst) const;
+  bool hasRewritableCandidates(const Instruction *Inst) const;
+  bool basisTooFarInSameBlock(
+      const Candidate &C,
+      DenseMap<const BasicBlock *, DenseMap<const Instruction *, int>>
+          &IndexCache,
+      const Instruction *Inst) const;
 };
 
 inline raw_ostream &operator<<(raw_ostream &OS,
@@ -1411,6 +1427,118 @@ bool StraightLineStrengthReduceLegacyPass::runOnFunction(Function &F) {
   return StraightLineStrengthReduce(DL, DT, SE, TTI).runOnFunction(F);
 }
 
+// Go through all operands of instruction, and check if any operand is used in
+// another block different from the instruction's block and the another use is
+// not rewritable. return true if such operand is found, otherwise return false.
+bool StraightLineStrengthReduce::
+    hasOperandsUsedInNonRewritableUsersInAnotherBlock(
+        llvm::Instruction *Inst) const {
+  llvm::BasicBlock *InstBB = Inst->getParent();
+  if (!InstBB)
+    return false;
+
+  for (Value *OpVal : Inst->operand_values()) {
+    auto *OpInst = dyn_cast<Instruction>(OpVal);
+    if (!OpInst)
+      continue;
+
+    for (const User *U : OpInst->users())
+      if (auto *UI = dyn_cast<Instruction>(U))
+        if (UI->getParent() != InstBB && !hasRewritableCandidates(UI)) {
+          LLVM_DEBUG(dbgs()
+                     << "Inst's operand is used in another block "
+                     << "("
+                     << (InstBB->hasName() ? InstBB->getName() : "unnamed")
+                     << " -> "
+                     << (UI->getParent() && UI->getParent()->hasName()
+                             ? UI->getParent()->getName()
+                             : "unnamed")
+                     << ") " << *UI << "\n");
+          return true;
+        }
+  }
+  return false;
+}
+
+bool StraightLineStrengthReduce::hasRewritableCandidates(
+    const Instruction *Inst) const {
+  if (!RewriteCandidates.count(Inst))
+    return false;
+
+  for (const Candidate *C : RewriteCandidates.at(Inst))
+    if (C->Basis)
+      return true;
+
+  return false;
+}
+
+// Assign a monotonically increasing index to each (non-debug/pseudo)
+// instruction in BB so in-block distances can be queried in O(1) once built.
+static DenseMap<const Instruction *, int>
+buildBlockIndexMap(const BasicBlock &BB) {
+  DenseMap<const Instruction *, int> IndexMap;
+  int Index = 0;
+  for (const Instruction &I : BB) {
+    // Skip debug/pseudo instructions so the distance math is identical
+    // between debug and release builds.
+    if (I.isDebugOrPseudoInst())
+      continue;
+    IndexMap[&I] = Index++;
+  }
+  return IndexMap;
+}
+
+// Return true
+// 1. if C.Basis used only in C.Ins's block before C.Ins
+// AND
+// 2. if any use of C.Basis before C.Ins and C.Ins exceeds
+// SLSRBasisDistanceThreshold.
+bool StraightLineStrengthReduce::basisTooFarInSameBlock(
+    const Candidate &C,
+    DenseMap<const BasicBlock *, DenseMap<const Instruction *, int>>
+        &IndexCache,
+    const Instruction *I) const {
+  Instruction *Inst = C.Ins;
+  assert(Inst == I);
+  Instruction *BasisInst = C.Basis ? C.Basis->Ins : nullptr;
+  if (!BasisInst)
+    return false;
+
+  const BasicBlock *BB = Inst->getParent();
+  auto [It, Inserted] = IndexCache.try_emplace(BB);
+  if (Inserted)
+    It->second = buildBlockIndexMap(*BB);
+  const DenseMap<const Instruction *, int> &IndexMap = It->second;
+
+  auto InstIt = IndexMap.find(Inst);
+  if (InstIt == IndexMap.end())
+    return false;
+  int InstIdx = InstIt->second;
+
+  int LastUseIdx = 0;
+
+  bool FoundSameBlockUse = false;
+  for (const User *U : BasisInst->users()) {
+    const auto *UI = dyn_cast<Instruction>(U);
+    if (!UI)
+      continue;
+    // If one of the uses is not in the same block, return false.
+    if (UI->getParent() != BB)
+      return false;
+    auto UseIt = IndexMap.find(UI);
+    // If any same block use is later than Inst, return false.
+    if (UseIt == IndexMap.end() || UseIt->second >= InstIdx)
+      return false;
+    FoundSameBlockUse = true;
+    LastUseIdx = std::max(LastUseIdx, UseIt->second);
+  }
+
+  if (!FoundSameBlockUse)
+    return false;
+
+  return (InstIdx - LastUseIdx) > SLSRBasisDistanceThreshold;
+}
+
 bool StraightLineStrengthReduce::runOnFunction(Function &F) {
   LLVM_DEBUG(dbgs() << "SLSR on Function: " << F.getName() << "\n");
   // Traverse the dominator tree in the depth-first order. This order makes sure
@@ -1427,11 +1555,50 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
   }
   sortCandidateInstructions();
 
+  // From SortedCandidateInsts, remove some candidates that are likely to
+  // increase register pressure. The candidate's Inst is the source of
+  // replacement. A candidate in the following criteria should be removed:
+  // 1. The candidate's Inst "has operands used in non-rewritable users in
+  // another block"
+  //    -- checked by hasOperandsUsedInNonRewritableUsersInAnotherBlock(Inst)
+  //    -- This means the candidate's Inst's original operands are live in
+  //    another block, so even if rewrite the Inst, the operands may be still
+  //    live out to another block.
+  //    -- Thus, rewriting the Inst based on Basis might add another long live
+  //    range from the Basis by increasing the live range of the Basis.
+  //    -- TODO: If needed, a refinement to check that "another block" is
+  //    properly dominated by the candidate's Inst's block can be added.
+  // 2. When the candidate's Basis's should beused only in the the same block
+  // and its last use is before the candidate's Inst, the difference between the
+  // last use of Basis and the Inst is larger than a threshold.
+  //    -- This is also for avoiding increasing the live range of the Basis by
+  //    rewriting the Inst.
+  //
+  // A candidate satisfies both conditions 1 and 2 should be removed.
+
+  // Collect candidates likely to increase register pressure.
+  // Evaluate on the original IR, before any rewriteCandidate mutates it
+  // Done before rewriting: rewriting inserts instructions and does
+  // replaceAllUsesWith, which would invalidate both the in-block index map and
+  // operands' user sets.
+  DenseSet<Instruction *> ToSkipRewrite;
+  {
+    DenseMap<const BasicBlock *, DenseMap<const Instruction *, int>> IndexCache;
+    for (Instruction *I : SortedCandidateInsts)
+      if (Candidate *C = pickRewriteCandidate(I))
+        if (hasOperandsUsedInNonRewritableUsersInAnotherBlock(I) &&
+            basisTooFarInSameBlock(*C, IndexCache, I))
+          ToSkipRewrite.insert(I);
+  }
+
   // Rewrite candidates in the topological order that rewrites a Candidate
   // always before rewriting its Basis
-  for (Instruction *I : reverse(SortedCandidateInsts))
+  for (Instruction *I : reverse(SortedCandidateInsts)) {
+    if (ToSkipRewrite.contains(I))
+      continue;
     if (Candidate *C = pickRewriteCandidate(I))
       rewriteCandidate(*C);
+  }
 
   for (auto *DeadIns : DeadInstructions)
     // A dead instruction may be another dead instruction's op,
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
new file mode 100644
index 0000000000000..96d05db79bd2e
--- /dev/null
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
@@ -0,0 +1,51 @@
+; RUN: opt < %s -passes=slsr -S -slsr-basis-distance-threshold=96 | FileCheck %s --check-prefixes=CHECK,REWRITE
+; RUN: opt < %s -passes=slsr -S -slsr-basis-distance-threshold=2  | FileCheck %s --check-prefixes=CHECK,SKIP
+
+; The register-pressure cost model skips a rewrite only when BOTH:
+;   1. an operand of the candidate is used by a non-rewritable user in
+;      another block, and
+;   2. the basis' last same-block use is farther than
+;      -slsr-basis-distance-threshold from the candidate.
+
+target datalayout = "e-i64:64-v16:16-v32:32-n16:32:64"
+
+declare void @foo(i32)
+declare void @use(i32)
+
+define void @basis_too_far(i32 %b, i32 %s) {
+; CHECK-LABEL: @basis_too_far(
+; CHECK:         %t1 = add i32 %b, %s
+; CHECK:         %s2 = shl i32 %s, 1
+; REWRITE:       %t2 = add i32 %t1, %s
+; SKIP:          %t2 = add i32 %b, %s2
+entry:
+  %t1 = add i32 %b, %s
+  call void @foo(i32 %t1)
+  call void @foo(i32 %b)
+  call void @foo(i32 %b)
+  call void @foo(i32 %b)
+  %s2 = shl i32 %s, 1
+  %t2 = add i32 %b, %s2
+  call void @foo(i32 %t2)
+  br label %next
+
+next:
+  call void @use(i32 %s2)
+  ret void
+}
+
+define void @same_block_operand(i32 %b, i32 %s) {
+; CHECK-LABEL: @same_block_operand(
+; CHECK:         %t2 = add i32 %t1, %s
+entry:
+  %t1 = add i32 %b, %s
+  call void @foo(i32 %t1)
+  call void @foo(i32 %b)
+  call void @foo(i32 %b)
+  call void @foo(i32 %b)
+  %s2 = shl i32 %s, 1
+  %t2 = add i32 %b, %s2
+  call void @foo(i32 %t2)
+  call void @use(i32 %s2)
+  ret void
+}

>From aceb0d71d8a235019c7bd23ab7bab79a39aa81bb Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Wed, 12 Aug 2026 06:38:47 +0000
Subject: [PATCH 02/13] Clean-ups based on reviewers' comments; Autogen a
 lit-test

---
 .../Scalar/StraightLineStrengthReduce.cpp     | 21 ++++---
 .../slsr-basis-distance-threshold.ll          | 57 ++++++++++++++++---
 2 files changed, 60 insertions(+), 18 deletions(-)

diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 31fe051cca640..abf14f10ad4ab 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -122,15 +122,13 @@ static cl::opt<bool>
     EnablePoisonReuseGuard("enable-poison-reuse-guard", cl::init(true),
                            cl::desc("Enable poison-reuse guard"));
 
-<<<<<<< HEAD
-STATISTIC(NumSCEVCandidateBasisDifferences,
-          "Number of candidate-basis SCEV differences computed by SLSR");
-=======
 static cl::opt<int> SLSRBasisDistanceThreshold(
     "slsr-basis-distance-threshold", cl::init(96), cl::Hidden,
     cl::desc("SLSR: skip rewrite if in-block distance from Basis's last "
              "same-block use to the candidate Inst exceeds this"));
->>>>>>> a5630b19d18c ([SLSR] Adding a cost model to avoid high regpressure)
+
+STATISTIC(NumSCEVCandidateBasisDifferences,
+          "Number of candidate-basis SCEV differences computed by SLSR");
 
 namespace {
 
@@ -1428,14 +1426,12 @@ bool StraightLineStrengthReduceLegacyPass::runOnFunction(Function &F) {
 }
 
 // Go through all operands of instruction, and check if any operand is used in
-// another block different from the instruction's block and the another use is
+// another block different from the instruction's block and the other use is
 // not rewritable. return true if such operand is found, otherwise return false.
 bool StraightLineStrengthReduce::
     hasOperandsUsedInNonRewritableUsersInAnotherBlock(
         llvm::Instruction *Inst) const {
   llvm::BasicBlock *InstBB = Inst->getParent();
-  if (!InstBB)
-    return false;
 
   for (Value *OpVal : Inst->operand_values()) {
     auto *OpInst = dyn_cast<Instruction>(OpVal);
@@ -1443,7 +1439,9 @@ bool StraightLineStrengthReduce::
       continue;
 
     for (const User *U : OpInst->users())
-      if (auto *UI = dyn_cast<Instruction>(U))
+      if (auto *UI = dyn_cast<Instruction>(U)) {
+        if (UI->isDebugOrPseudoInst())
+          continue;
         if (UI->getParent() != InstBB && !hasRewritableCandidates(UI)) {
           LLVM_DEBUG(dbgs()
                      << "Inst's operand is used in another block "
@@ -1456,6 +1454,7 @@ bool StraightLineStrengthReduce::
                      << ") " << *UI << "\n");
           return true;
         }
+      }
   }
   return false;
 }
@@ -1543,7 +1542,7 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
   LLVM_DEBUG(dbgs() << "SLSR on Function: " << F.getName() << "\n");
   // Traverse the dominator tree in the depth-first order. This order makes sure
   // all bases of a candidate are in Candidates when we process it.
-  for (const auto Node : depth_first(DT))
+  for (auto *const Node : depth_first(DT))
     for (auto &I : *(Node->getBlock()))
       allocateCandidatesAndFindBasis(&I);
 
@@ -1568,7 +1567,7 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
   //    range from the Basis by increasing the live range of the Basis.
   //    -- TODO: If needed, a refinement to check that "another block" is
   //    properly dominated by the candidate's Inst's block can be added.
-  // 2. When the candidate's Basis's should beused only in the the same block
+  // 2. When the candidate's Basis's is only used in the the same block
   // and its last use is before the candidate's Inst, the difference between the
   // last use of Basis and the Inst is larger than a threshold.
   //    -- This is also for avoiding increasing the live range of the Basis by
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
index 96d05db79bd2e..42aa95b8dd006 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
@@ -1,3 +1,4 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; RUN: opt < %s -passes=slsr -S -slsr-basis-distance-threshold=96 | FileCheck %s --check-prefixes=CHECK,REWRITE
 ; RUN: opt < %s -passes=slsr -S -slsr-basis-distance-threshold=2  | FileCheck %s --check-prefixes=CHECK,SKIP
 
@@ -6,6 +7,9 @@
 ;      another block, and
 ;   2. the basis' last same-block use is farther than
 ;      -slsr-basis-distance-threshold from the candidate.
+;
+; Rewriting of %t2 = add i32 %t1, %s based on basis %t1 shouldn't happen when slsr-basis-distance-threshold is 2.
+; Distance from %t1's last use to %t2 is 5 > 2 and %s2 continues to be used in "next".
 
 target datalayout = "e-i64:64-v16:16-v32:32-n16:32:64"
 
@@ -13,11 +17,38 @@ declare void @foo(i32)
 declare void @use(i32)
 
 define void @basis_too_far(i32 %b, i32 %s) {
-; CHECK-LABEL: @basis_too_far(
-; CHECK:         %t1 = add i32 %b, %s
-; CHECK:         %s2 = shl i32 %s, 1
-; REWRITE:       %t2 = add i32 %t1, %s
-; SKIP:          %t2 = add i32 %b, %s2
+; REWRITE-LABEL: define void @basis_too_far(
+; REWRITE-SAME: i32 [[B:%.*]], i32 [[S:%.*]]) {
+; REWRITE-NEXT:  [[ENTRY:.*:]]
+; REWRITE-NEXT:    [[T1:%.*]] = add i32 [[B]], [[S]]
+; REWRITE-NEXT:    call void @foo(i32 [[T1]])
+; REWRITE-NEXT:    call void @foo(i32 [[B]])
+; REWRITE-NEXT:    call void @foo(i32 [[B]])
+; REWRITE-NEXT:    call void @foo(i32 [[B]])
+; REWRITE-NEXT:    [[S2:%.*]] = shl i32 [[S]], 1
+; REWRITE-NEXT:    [[T2:%.*]] = add i32 [[T1]], [[S]]
+; REWRITE-NEXT:    call void @foo(i32 [[T2]])
+; REWRITE-NEXT:    br label %[[NEXT:.*]]
+; REWRITE:       [[NEXT]]:
+; REWRITE-NEXT:    call void @use(i32 [[S2]])
+; REWRITE-NEXT:    ret void
+;
+; SKIP-LABEL: define void @basis_too_far(
+; SKIP-SAME: i32 [[B:%.*]], i32 [[S:%.*]]) {
+; SKIP-NEXT:  [[ENTRY:.*:]]
+; SKIP-NEXT:    [[T1:%.*]] = add i32 [[B]], [[S]]
+; SKIP-NEXT:    call void @foo(i32 [[T1]])
+; SKIP-NEXT:    call void @foo(i32 [[B]])
+; SKIP-NEXT:    call void @foo(i32 [[B]])
+; SKIP-NEXT:    call void @foo(i32 [[B]])
+; SKIP-NEXT:    [[S2:%.*]] = shl i32 [[S]], 1
+; SKIP-NEXT:    [[T2:%.*]] = add i32 [[B]], [[S2]]
+; SKIP-NEXT:    call void @foo(i32 [[T2]])
+; SKIP-NEXT:    br label %[[NEXT:.*]]
+; SKIP:       [[NEXT]]:
+; SKIP-NEXT:    call void @use(i32 [[S2]])
+; SKIP-NEXT:    ret void
+;
 entry:
   %t1 = add i32 %b, %s
   call void @foo(i32 %t1)
@@ -35,8 +66,20 @@ next:
 }
 
 define void @same_block_operand(i32 %b, i32 %s) {
-; CHECK-LABEL: @same_block_operand(
-; CHECK:         %t2 = add i32 %t1, %s
+; CHECK-LABEL: define void @same_block_operand(
+; CHECK-SAME: i32 [[B:%.*]], i32 [[S:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[T1:%.*]] = add i32 [[B]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T1]])
+; CHECK-NEXT:    call void @foo(i32 [[B]])
+; CHECK-NEXT:    call void @foo(i32 [[B]])
+; CHECK-NEXT:    call void @foo(i32 [[B]])
+; CHECK-NEXT:    [[S2:%.*]] = shl i32 [[S]], 1
+; CHECK-NEXT:    [[T2:%.*]] = add i32 [[T1]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T2]])
+; CHECK-NEXT:    call void @use(i32 [[S2]])
+; CHECK-NEXT:    ret void
+;
 entry:
   %t1 = add i32 %b, %s
   call void @foo(i32 %t1)

>From cd17f6cf22f0b932e903d77bd735317392170b44 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Sat, 22 Aug 2026 22:08:44 +0000
Subject: [PATCH 03/13] Liveness-based RPFilter

---
 .../Scalar/StraightLineStrengthReduce.cpp     | 411 ++++++++++++++++++
 1 file changed, 411 insertions(+)

diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index abf14f10ad4ab..da75915ec3e94 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -71,6 +71,7 @@
 #include "llvm/Transforms/Scalar/StraightLineStrengthReduce.h"
 #include "llvm/ADT/APInt.h"
 #include "llvm/ADT/DepthFirstIterator.h"
+#include "llvm/ADT/PostOrderIterator.h"
 #include "llvm/ADT/SetVector.h"
 #include "llvm/ADT/SmallPtrSet.h"
 #include "llvm/ADT/SmallVector.h"
@@ -110,6 +111,7 @@ using namespace llvm;
 using namespace PatternMatch;
 
 #define DEBUG_TYPE "slsr"
+#define DEBUG_SLSR_RP(X) DEBUG_WITH_TYPE(DEBUG_TYPE "-rp", X)
 
 static const unsigned UnknownAddressSpace =
     std::numeric_limits<unsigned>::max();
@@ -1538,6 +1540,395 @@ bool StraightLineStrengthReduce::basisTooFarInSameBlock(
   return (InstIdx - LastUseIdx) > SLSRBasisDistanceThreshold;
 }
 
+namespace {
+
+// TODO: Currently, I am considering (Basis, Cand) pair that are both in the
+// same BB.
+//       The restriction may not be needed.
+class RPFilter {
+public:
+  using Candidate = StraightLineStrengthReduce::Candidate;
+
+  RPFilter(const Function *F,
+           DenseMap<Instruction *, Candidate *> &PickedCandidateMap)
+      //, const TargetTransformInfo *TTI)
+      : F(F), PickedCandidateMap(PickedCandidateMap) {} //, TTI(TTI) {}
+  void run() {
+    buildBBToNumCandsAndBasises(PickedCandidateMap);
+    // TODO: 16 is an arbitrary threshold.
+    if (MaxNumBasisesInBB > 16)
+      buildBBToLiveness(*F); // Do liveness analysis.
+
+    for (auto &BB : *F) {
+#if 0
+      unsigned RP = maxPressureInBlock(BB, BBToLiveness[&BB].LiveIn,
+                                       BBToLiveness[&BB].LiveOut);
+      unsigned RPBackward = maxPressureInBlockBackward(BB, BBToLiveness[&BB].LiveIn,
+                                      BBToLiveness[&BB].LiveOut);
+      DEBUG_WITH_TYPE("slsr-rp", {
+        dbgs() << "MaxRP:" << BB.getName() << ":" << RP << "," << RPBackward << "\n";
+      });
+#else
+      unsigned RPBackward = maxPressureInBlockBackward(
+          BB, BBToLiveness[&BB].LiveIn, BBToLiveness[&BB].LiveOut);
+      DEBUG_SLSR_RP(dbgs() << "MaxRPBackward:" << BB.getName() << ":"
+                           << RPBackward << "\n");
+#endif
+    }
+  }
+
+private:
+  const Function *F;
+  DenseMap<Instruction *, Candidate *> &PickedCandidateMap;
+  // const TargetTransformInfo *TTI;
+
+  DenseMap<const BasicBlock *, std::pair<unsigned, unsigned>>
+      BBToNumCandsAndBasises;
+  unsigned MaxNumBasisesInBB = 0;
+
+  using ValueSet = SmallPtrSet<const Value *, 32>;
+  struct BlockLiveness {
+    ValueSet LiveIn;
+    ValueSet LiveOut;
+  };
+
+  DenseMap<const BasicBlock *, BlockLiveness> BBToLiveness;
+
+  std::pair<unsigned, unsigned> countCandsAndBasisesInBB(
+      const BasicBlock *BB,
+      const DenseMap<Instruction *, Candidate *> &PickedCandidateMap) {
+    unsigned NumCands = 0;
+    SmallPtrSet<const Candidate *, 8> UniqueBasises;
+    for (const Instruction &Inst : *BB) {
+      auto It = PickedCandidateMap.find(&Inst);
+      if (It != PickedCandidateMap.end()) {
+        NumCands++;
+        assert(It->second->Basis);
+        UniqueBasises.insert(It->second->Basis);
+      }
+    }
+    MaxNumBasisesInBB = std::max(MaxNumBasisesInBB, UniqueBasises.size());
+    DEBUG_WITH_TYPE("slsr-rp", {
+      dbgs() << "BB: " << BB->getName() << " NumCands: " << NumCands
+             << " UniqueBasises: " << UniqueBasises.size() << "\n";
+    });
+    return {NumCands, UniqueBasises.size()};
+  }
+
+  void buildBBToNumCandsAndBasises(
+      const DenseMap<Instruction *, Candidate *> &PickedCandidateMap) {
+    for (auto &InstCand : PickedCandidateMap) {
+      const BasicBlock *BB = InstCand.first->getParent();
+      auto [It, Inserted] = BBToNumCandsAndBasises.try_emplace(BB);
+      if (Inserted)
+        It->second = countCandsAndBasisesInBB(It->first, PickedCandidateMap);
+    }
+  }
+
+  static bool isRegisterLike(const Value *V) {
+    return isa<Instruction>(V) || isa<Argument>(V);
+  }
+
+  unsigned regWeight(const Value *V) const {
+    const DataLayout &DL = F->getDataLayout();
+    Type *Ty = V->getType();
+    if (Ty->isVoidTy() || Ty->isTokenTy())
+      return 0;
+#if 0
+    // Aggregates legalize to a flat sequence of scalars; approximate rather
+    // than calling getRegUsageForType, which is llvm_unreachable on them.
+    if (!VectorType::isValidElementType(Ty->getScalarType()))
+      return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
+    return TTI->getRegUsageForType(Ty);
+#else
+    // TTI's getRegUsageForType can be used for types that are
+    // validElementType(Ty->getScalarType()). However, even the valid vector
+    // type should be multiplited by get{Min}NumElements() to handle
+    // {ScalarVectorType}, FixedVectorType. Overall, getTypeSizeInBits(Ty)
+    // handing all those cases should be sufficient for heuristic.
+    return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
+#endif
+  }
+
+  void buildBBToLiveness(const Function &F) {
+    DenseMap<const BasicBlock *, ValueSet> UpExposed, Defs;
+    DenseMap<std::pair<const BasicBlock *, const BasicBlock *>, ValueSet>
+        PhiEdges;
+
+    // Fill in per-block information
+    for (const BasicBlock &BB : F) {
+      ValueSet &UE = UpExposed[&BB];
+      ValueSet &D = Defs[&BB];
+
+      for (const Instruction &I : BB) {
+        if (const auto *PN = dyn_cast<PHINode>(&I)) {
+          for (unsigned i = 0, e = PN->getNumIncomingValues(); i < e; ++i)
+            if (isRegisterLike(PN->getIncomingValue(i)))
+              PhiEdges[{PN->getIncomingBlock(i), &BB}].insert(
+                  PN->getIncomingValue(i));
+        } else {
+          for (const Value *Op : I.operand_values())
+            if (isRegisterLike(Op) && !D.contains(Op))
+              UE.insert(Op);
+        }
+
+        if (!I.getType()->isVoidTy())
+          D.insert(&I);
+      }
+      DEBUG_WITH_TYPE("slsr-rp", {
+        dbgs() << "BB-fill: " << BB.getName()
+               << " UpExposed: " << UpExposed[&BB].size();
+        dbgs() << " Defs: " << Defs[&BB].size() << "\n";
+      });
+    }
+
+    // Block Live-in/out fixed-point loop
+    bool Changed = true;
+    while (Changed) {
+      Changed = false;
+      for (const BasicBlock *BB : post_order(&F.getEntryBlock())) {
+        ValueSet Out;
+        for (const BasicBlock *S : successors(BB)) {
+          // Out = Out + BBToLiveness[S].LiveIn
+          for (const Value *V : BBToLiveness[S].LiveIn)
+            Out.insert(V);
+          auto It = PhiEdges.find({BB, S});
+          if (It != PhiEdges.end())
+            // Out = Out + It->second
+            for (const Value *V : It->second)
+              Out.insert(V);
+        }
+        ValueSet In = UpExposed[BB];
+        // live-throughs are live-ins
+        for (const Value *V : Out)
+          if (!Defs[BB].contains(V))
+            In.insert(V);
+
+        if (In.size() != BBToLiveness[BB].LiveIn.size() ||
+            Out.size() != BBToLiveness[BB].LiveOut.size())
+          Changed = true;
+
+        DEBUG_WITH_TYPE("slsr-rp", {
+          dbgs() << "BB-update: " << BB->getName() << " In: " << In.size()
+                 << " Out: " << Out.size() << "\n";
+        });
+
+        BBToLiveness[BB].LiveIn = std::move(In);
+        BBToLiveness[BB].LiveOut = std::move(Out);
+      }
+    }
+
+    DEBUG_WITH_TYPE("slsr-rp", {
+      dbgs() << "-- Liveness of BBs -- \n";
+      for (const BasicBlock &BB : F) {
+        dbgs() << BB.getName() << ": ";
+        dbgs() << BBToLiveness[&BB].LiveIn.size() << ", ";
+        dbgs() << BBToLiveness[&BB].LiveOut.size() << "\n";
+      }
+    });
+  }
+
+  bool isDebugBlock(const BasicBlock &BB) const {
+    return BB.getName() == "for.cond.cleanup";
+  }
+
+  unsigned maxPressureInBlock(const BasicBlock &BB, const ValueSet &LiveIn,
+                              const ValueSet &LiveOut) const {
+
+#if 1
+    bool IsDebugBlock = isDebugBlock(BB);
+    unsigned StoreCount = 0;
+#endif
+
+    SmallVector<const Instruction *, 128> Order;
+    DenseMap<const Instruction *, unsigned> Idx;
+    for (const Instruction &I : BB) {
+      if (I.isDebugOrPseudoInst())
+        continue;
+      Idx[&I] = Order.size();
+      Order.push_back(&I);
+    }
+
+    DenseMap<const Value *, unsigned> LastUse;
+    for (const Instruction &I : BB) {
+      if (I.isDebugOrPseudoInst() || isa<PHINode>(&I))
+        continue;
+      for (const Value *Op : I.operand_values())
+        if (isRegisterLike(Op))
+          LastUse[Op] = Idx[&I];
+    }
+    const unsigned End = Order.size();
+    for (const Value *V : LiveOut)
+      LastUse[V] = End; // survives the block; never retires here
+
+    ValueSet Open;
+    for (const Value *V : LiveIn)
+      Open.insert(V);
+
+    unsigned MaxW = 0;
+    for (unsigned i = 0; i != End; ++i) {
+      if (isa<StoreInst>(Order[i])) {
+        StoreCount++;
+      }
+      if (!Order[i]->getType()->isVoidTy())
+        Open.insert(Order[i]);
+
+      unsigned W = 0;
+      for (const Value *V : Open) {
+        auto RW = regWeight(V);
+        W += RW;
+        if (IsDebugBlock && StoreCount == 32) {
+          DEBUG_SLSR_RP({
+            dbgs() << "regweight: " << "i: " << i << " value: " << *V
+                   << " RW: " << RW << ", ";
+          });
+          // dbgs() << "W: " << W << "\n";
+        }
+
+        // W += regWeight(V);
+      }
+      MaxW = std::max(MaxW, W);
+
+      SmallVector<const Value *, 8> Dead;
+      for (const Value *V : Open) {
+        auto It = LastUse.find(V);
+        if (It == LastUse.end() || It->second <= i)
+          Dead.push_back(V);
+      }
+      for (const Value *V : Dead)
+        Open.erase(V);
+
+      if (IsDebugBlock && StoreCount == 32) {
+        DEBUG_SLSR_RP(dbgs() << "W: " << W << " MaxW: " << MaxW << "\n");
+      }
+    }
+    return MaxW;
+  }
+
+  // Return true if I is a candidate and its basis is in the same bb, false
+  // otherwise.
+  bool insertBasisIfCand(const Instruction *I, ValueSet &LiveSetWithSLSR,
+                         DenseSet<const Value *> &SeenLastUse) const {
+    auto It = PickedCandidateMap.find(I);
+    if (It == PickedCandidateMap.end())
+      return false;
+
+    // I is a Candidate
+    const Instruction *Basis = It->second->Basis->Ins;
+    if (Basis->getParent() == I->getParent()) {
+      LiveSetWithSLSR.insert(Basis);
+      // I and its Basis are in the same bb.
+
+      SeenLastUse.insert(Basis);
+      return true;
+    }
+
+    return false;
+  }
+
+  // TODO: Remove
+  void updateLiveSetWithSLSR(ValueSet &LiveSetWithSLSR,
+                             const DenseSet<const Value *> &SeenLastUse,
+                             const Instruction &I, const Value *Op) const {
+    // LiveSet += {Cand.Basis} <-- Done already
+    // LiveSet -= {Op} if this is the last use of Op (i.e.
+    // SeenLastUse.contains(Op))
+    if (SeenLastUse.contains(Op))
+      LiveSetWithSLSR.erase(Op);
+  }
+
+  unsigned maxPressureInBlockBackward(const BasicBlock &BB,
+                                      const ValueSet &LiveIn,
+                                      const ValueSet &LiveOut) const {
+
+    // TODO: ValueSet is a SmallPtrSet, which is supposedly smaller than 33.
+    //       Could be a better-fitting data structure.
+    ValueSet LiveSet = LiveOut;
+    unsigned MaxW = 0;
+    for (const Value *V : LiveSet)
+      MaxW += regWeight(V);
+
+    // Initial LiveSetWithSLSR is the same as LiveOut.
+    // SLSR changes
+    // Basis = ..
+    // ..
+    // Cand = f(op1, op2, ..)
+    //
+    // to
+    // Basis = ..
+    // Cand = f(Basis, Delta, // possibly some of the original operands ..)
+    //
+    // We consider here only the case both Cand and Basis are in the same BB.
+    // Therefore, if an original operand of Cand is a liveout, it means it was
+    // used outside the BB, and will stay so. Likewise, if a Basis is a liveout,
+    // it means it was used outside the BB, and will stay so. SLSR does not
+    // delete existing defs or create new defs.
+    ValueSet LiveSetWithSLSR = LiveOut;
+    unsigned MaxWWithSLSR = MaxW;
+
+    // This is required for calculating LiveSetWithSLSR without updating IR.
+    // IR is still in the original form.
+    // Intialize it with LiveOut. LiveOuts are never removed from
+    // LiveSetWithSLSR.
+    DenseSet<const Value *> SeenLastUse(LiveOut.begin(), LiveOut.end());
+
+    // Scan BB backward
+    for (const Instruction &I : reverse(BB)) {
+      if (I.isDebugOrPseudoInst() || isa<PHINode>(&I))
+        continue;
+
+      bool IsCand = insertBasisIfCand(&I, LiveSetWithSLSR, SeenLastUse);
+      for (const Value *Op : I.operand_values())
+        if (isRegisterLike(Op)) {
+          LiveSet.insert(Op);
+          if (IsCand && !SeenLastUse.contains(Op)) {
+            // Op is the last use. It will be replaced by Basis, so it isremoved
+            // from LiveSetWithSLSR. As a heuristic, we don't dicern which Op of
+            // I is replaced by Basis. When IsCand is true, there is only
+            // reg-like one Op in practice.
+            // TODO: Can further restrict by Candidate's type and DeltaKind.
+            LiveSetWithSLSR.erase(Op);
+            DEBUG_SLSR_RP(dbgs() << "Removed Op from LiveSetWithSLSR: " << *Op
+                                 << " in inst " << I << "\n");
+          } else {
+            LiveSetWithSLSR.insert(Op);
+          }
+          // Op's insertion happens only if Op was not there before.
+          // This logic works as BB is scanned backward.
+          SeenLastUse.insert(Op);
+        }
+
+      // Compute RP reaching for this instruction.
+      unsigned W = 0;
+      for (const Value *V : LiveSet) {
+        // TODO: Cache regWeight(V)
+        auto RW = regWeight(V);
+        W += RW;
+      }
+      MaxW = std::max(MaxW, W);
+      W = 0;
+      for (const Value *V : LiveSetWithSLSR) {
+        // TODO: Cache regWeight(V)
+        auto RW = regWeight(V);
+        W += RW;
+      }
+      MaxWWithSLSR = std::max(MaxWWithSLSR, W);
+
+      // Remove the def (Instruction) from LiveSet for the next upward
+      // instruction.
+      if (!I.getType()->isVoidTy()) {
+        LiveSet.erase(&I);
+        LiveSetWithSLSR.erase(&I);
+      }
+    }
+
+    DEBUG_SLSR_RP(dbgs() << "MaxWWithSLSR: " << MaxWWithSLSR << "\n");
+
+    return MaxW;
+  }
+};
+} // end of anonymous namespace
+
 bool StraightLineStrengthReduce::runOnFunction(Function &F) {
   LLVM_DEBUG(dbgs() << "SLSR on Function: " << F.getName() << "\n");
   // Traverse the dominator tree in the depth-first order. This order makes sure
@@ -1554,6 +1945,26 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
   }
   sortCandidateInstructions();
 
+  ////////////////////////////////////////////////////
+  DenseMap<Instruction *, Candidate *> PickedCandidateMap;
+  for (Instruction *I : SortedCandidateInsts)
+    if (Candidate *C = pickRewriteCandidate(I))
+      PickedCandidateMap[I] = C;
+
+  // RPFilter RPFilter(&F, PickedCandidateMap, TTI);
+  RPFilter RPFilter(&F, PickedCandidateMap);
+  RPFilter.run();
+
+#if 0
+  DenseMap<const BasicBlock*, unsigned> BBToNumCands;
+  for (auto &InstCand : PickedCandidateMap) {
+    const BasicBlock *BB= InstCand.first->getParent();
+    auto [It, Inserted] = BBToNumCands.try_emplace(BB);
+    if (Inserted)
+      It->second = countCandsInBB(It->first, PickedCandidateMap);
+  }
+#endif
+
   // From SortedCandidateInsts, remove some candidates that are likely to
   // increase register pressure. The candidate's Inst is the source of
   // replacement. A candidate in the following criteria should be removed:

>From e6a11a1f824899f6bbd1d8be4535d13362a57793 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Tue, 25 Aug 2026 00:10:03 +0000
Subject: [PATCH 04/13] Adding an option to use TTI's getRegUsageFor as an
 alternative

---
 .../Scalar/StraightLineStrengthReduce.cpp     | 90 +++++++++++--------
 1 file changed, 52 insertions(+), 38 deletions(-)

diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index da75915ec3e94..8fb8dced13072 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -112,6 +112,7 @@ using namespace PatternMatch;
 
 #define DEBUG_TYPE "slsr"
 #define DEBUG_SLSR_RP(X) DEBUG_WITH_TYPE(DEBUG_TYPE "-rp", X)
+#define DEBUG_SLSR_RP_DETAIL(X) DEBUG_WITH_TYPE(DEBUG_TYPE "-rp-detail", X)
 
 static const unsigned UnknownAddressSpace =
     std::numeric_limits<unsigned>::max();
@@ -129,6 +130,9 @@ static cl::opt<int> SLSRBasisDistanceThreshold(
     cl::desc("SLSR: skip rewrite if in-block distance from Basis's last "
              "same-block use to the candidate Inst exceeds this"));
 
+static cl::opt<bool> UseTTIForRP("slsr-tti-rp", cl::init(false),
+                                 cl::desc("Use TTI to compute RP for SLSR-RP"));
+
 STATISTIC(NumSCEVCandidateBasisDifferences,
           "Number of candidate-basis SCEV differences computed by SLSR");
 
@@ -1550,15 +1554,17 @@ class RPFilter {
   using Candidate = StraightLineStrengthReduce::Candidate;
 
   RPFilter(const Function *F,
-           DenseMap<Instruction *, Candidate *> &PickedCandidateMap)
-      //, const TargetTransformInfo *TTI)
-      : F(F), PickedCandidateMap(PickedCandidateMap) {} //, TTI(TTI) {}
+           DenseMap<Instruction *, Candidate *> &PickedCandidateMap,
+           const TargetTransformInfo *TTI)
+      : F(F), PickedCandidateMap(PickedCandidateMap), TTI(TTI) {}
+
   void run() {
     buildBBToNumCandsAndBasises(PickedCandidateMap);
     // TODO: 16 is an arbitrary threshold.
     if (MaxNumBasisesInBB > 16)
       buildBBToLiveness(*F); // Do liveness analysis.
 
+    DEBUG_SLSR_RP(dbgs() << "-- MaxRP of BBs -- \n");
     for (auto &BB : *F) {
 #if 0
       unsigned RP = maxPressureInBlock(BB, BBToLiveness[&BB].LiveIn,
@@ -1569,10 +1575,10 @@ class RPFilter {
         dbgs() << "MaxRP:" << BB.getName() << ":" << RP << "," << RPBackward << "\n";
       });
 #else
-      unsigned RPBackward = maxPressureInBlockBackward(
+      auto [MaxRP, MaxRPWithSLSR] = maxPressureInBlockBackward(
           BB, BBToLiveness[&BB].LiveIn, BBToLiveness[&BB].LiveOut);
-      DEBUG_SLSR_RP(dbgs() << "MaxRPBackward:" << BB.getName() << ":"
-                           << RPBackward << "\n");
+      DEBUG_SLSR_RP(dbgs() << "MaxRP:" << BB.getName() << ": (" << MaxRP << ", "
+                           << MaxRPWithSLSR << ")" << "\n");
 #endif
     }
   }
@@ -1580,7 +1586,7 @@ class RPFilter {
 private:
   const Function *F;
   DenseMap<Instruction *, Candidate *> &PickedCandidateMap;
-  // const TargetTransformInfo *TTI;
+  const TargetTransformInfo *TTI;
 
   DenseMap<const BasicBlock *, std::pair<unsigned, unsigned>>
       BBToNumCandsAndBasises;
@@ -1609,7 +1615,7 @@ class RPFilter {
     }
     MaxNumBasisesInBB = std::max(MaxNumBasisesInBB, UniqueBasises.size());
     DEBUG_WITH_TYPE("slsr-rp", {
-      dbgs() << "BB: " << BB->getName() << " NumCands: " << NumCands
+      dbgs() << "BB: " << BB->getName() << " - NumCands: " << NumCands
              << " UniqueBasises: " << UniqueBasises.size() << "\n";
     });
     return {NumCands, UniqueBasises.size()};
@@ -1634,20 +1640,30 @@ class RPFilter {
     Type *Ty = V->getType();
     if (Ty->isVoidTy() || Ty->isTokenTy())
       return 0;
+    if (UseTTIForRP) {
+      // Aggregates legalize to a flat sequence of scalars; approximate rather
+      // than calling getRegUsageForType, which is llvm_unreachable on them.
+      if (!VectorType::isValidElementType(Ty->getScalarType()))
+        return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
+
+      // TTI's getRegUsageForType can be used for types that are
+      // validElementType(Ty->getScalarType()). However, even the valid vector
+      // type should be multiplited by get{Min}NumElements() to handle
+      // {ScalarVectorType}, FixedVectorType. Overall, getTypeSizeInBits(Ty)
+      // handing all those cases should be sufficient for heuristic.
+      unsigned RegUsage = TTI->getRegUsageForType(Ty);
 #if 0
-    // Aggregates legalize to a flat sequence of scalars; approximate rather
-    // than calling getRegUsageForType, which is llvm_unreachable on them.
-    if (!VectorType::isValidElementType(Ty->getScalarType()))
-      return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
-    return TTI->getRegUsageForType(Ty);
-#else
-    // TTI's getRegUsageForType can be used for types that are
-    // validElementType(Ty->getScalarType()). However, even the valid vector
-    // type should be multiplited by get{Min}NumElements() to handle
-    // {ScalarVectorType}, FixedVectorType. Overall, getTypeSizeInBits(Ty)
-    // handing all those cases should be sufficient for heuristic.
-    return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
+      if (FixedVectorType *FVT = dyn_cast<FixedVectorType>(Ty))
+        RegUsage *= FVT->getNumElements();
+      else if (ScalableVectorType *SVT = dyn_cast<ScalableVectorType>(Ty))
+        RegUsage *= SVT->getMinNumElements();
 #endif
+      return RegUsage;
+
+    } else {
+      // Default logic to compute RP.
+      return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
+    }
   }
 
   void buildBBToLiveness(const Function &F) {
@@ -1675,7 +1691,7 @@ class RPFilter {
         if (!I.getType()->isVoidTy())
           D.insert(&I);
       }
-      DEBUG_WITH_TYPE("slsr-rp", {
+      DEBUG_SLSR_RP_DETAIL({
         dbgs() << "BB-fill: " << BB.getName()
                << " UpExposed: " << UpExposed[&BB].size();
         dbgs() << " Defs: " << Defs[&BB].size() << "\n";
@@ -1708,7 +1724,7 @@ class RPFilter {
             Out.size() != BBToLiveness[BB].LiveOut.size())
           Changed = true;
 
-        DEBUG_WITH_TYPE("slsr-rp", {
+        DEBUG_SLSR_RP_DETAIL({
           dbgs() << "BB-update: " << BB->getName() << " In: " << In.size()
                  << " Out: " << Out.size() << "\n";
         });
@@ -1719,7 +1735,7 @@ class RPFilter {
     }
 
     DEBUG_WITH_TYPE("slsr-rp", {
-      dbgs() << "-- Liveness of BBs -- \n";
+      dbgs() << "-- Live Ins/Outs of BBs -- \n";
       for (const BasicBlock &BB : F) {
         dbgs() << BB.getName() << ": ";
         dbgs() << BBToLiveness[&BB].LiveIn.size() << ", ";
@@ -1837,9 +1853,9 @@ class RPFilter {
       LiveSetWithSLSR.erase(Op);
   }
 
-  unsigned maxPressureInBlockBackward(const BasicBlock &BB,
-                                      const ValueSet &LiveIn,
-                                      const ValueSet &LiveOut) const {
+  std::pair<unsigned, unsigned>
+  maxPressureInBlockBackward(const BasicBlock &BB, const ValueSet &LiveIn,
+                             const ValueSet &LiveOut) const {
 
     // TODO: ValueSet is a SmallPtrSet, which is supposedly smaller than 33.
     //       Could be a better-fitting data structure.
@@ -1882,19 +1898,20 @@ class RPFilter {
         if (isRegisterLike(Op)) {
           LiveSet.insert(Op);
           if (IsCand && !SeenLastUse.contains(Op)) {
-            // Op is the last use. It will be replaced by Basis, so it isremoved
-            // from LiveSetWithSLSR. As a heuristic, we don't dicern which Op of
-            // I is replaced by Basis. When IsCand is true, there is only
-            // reg-like one Op in practice.
+            // Op is the last use. It will be replaced by Basis, so it is
+            // removed from LiveSetWithSLSR. As a heuristic, we don't dicern
+            // which Op of I is replaced by Basis. When IsCand is true, there is
+            // only one reg-like Op in practice.
             // TODO: Can further restrict by Candidate's type and DeltaKind.
             LiveSetWithSLSR.erase(Op);
-            DEBUG_SLSR_RP(dbgs() << "Removed Op from LiveSetWithSLSR: " << *Op
-                                 << " in inst " << I << "\n");
+            DEBUG_SLSR_RP_DETAIL(dbgs() << "Removed Op from LiveSetWithSLSR: "
+                                        << *Op << " in inst " << I << "\n");
           } else {
             LiveSetWithSLSR.insert(Op);
           }
           // Op's insertion happens only if Op was not there before.
-          // This logic works as BB is scanned backward.
+          // Therefore, SeenLastUse keeps the last use of Op as
+          // BB is scanned backward.
           SeenLastUse.insert(Op);
         }
 
@@ -1922,9 +1939,7 @@ class RPFilter {
       }
     }
 
-    DEBUG_SLSR_RP(dbgs() << "MaxWWithSLSR: " << MaxWWithSLSR << "\n");
-
-    return MaxW;
+    return {MaxW, MaxWWithSLSR};
   }
 };
 } // end of anonymous namespace
@@ -1951,8 +1966,7 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
     if (Candidate *C = pickRewriteCandidate(I))
       PickedCandidateMap[I] = C;
 
-  // RPFilter RPFilter(&F, PickedCandidateMap, TTI);
-  RPFilter RPFilter(&F, PickedCandidateMap);
+  RPFilter RPFilter(&F, PickedCandidateMap, TTI);
   RPFilter.run();
 
 #if 0

>From ee0a71e60f35e897b345104c1f06a807550aaf0f Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Tue, 25 Aug 2026 01:27:55 +0000
Subject: [PATCH 05/13] Add cache for regWeight

---
 .../Scalar/StraightLineStrengthReduce.cpp     | 33 +++++++++----------
 1 file changed, 16 insertions(+), 17 deletions(-)

diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 8fb8dced13072..20a309c0091a9 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -1635,11 +1635,23 @@ class RPFilter {
     return isa<Instruction>(V) || isa<Argument>(V);
   }
 
+  // The pressure scan asks for the weight of every live value at every
+  // instruction, so memoize on the type, which is all the weight depends on.
+  mutable DenseMap<Type *, unsigned> RegWeightCache;
+
   unsigned regWeight(const Value *V) const {
-    const DataLayout &DL = F->getDataLayout();
     Type *Ty = V->getType();
     if (Ty->isVoidTy() || Ty->isTokenTy())
       return 0;
+
+    auto [It, Inserted] = RegWeightCache.try_emplace(Ty);
+    if (Inserted)
+      It->second = computeRegWeight(Ty);
+    return It->second;
+  }
+
+  unsigned computeRegWeight(Type *Ty) const {
+    const DataLayout &DL = F->getDataLayout();
     if (UseTTIForRP) {
       // Aggregates legalize to a flat sequence of scalars; approximate rather
       // than calling getRegUsageForType, which is llvm_unreachable on them.
@@ -1659,11 +1671,10 @@ class RPFilter {
         RegUsage *= SVT->getMinNumElements();
 #endif
       return RegUsage;
-
-    } else {
-      // Default logic to compute RP.
-      return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
     }
+
+    // Default logic to compute RP.
+    return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
   }
 
   void buildBBToLiveness(const Function &F) {
@@ -1918,14 +1929,12 @@ class RPFilter {
       // Compute RP reaching for this instruction.
       unsigned W = 0;
       for (const Value *V : LiveSet) {
-        // TODO: Cache regWeight(V)
         auto RW = regWeight(V);
         W += RW;
       }
       MaxW = std::max(MaxW, W);
       W = 0;
       for (const Value *V : LiveSetWithSLSR) {
-        // TODO: Cache regWeight(V)
         auto RW = regWeight(V);
         W += RW;
       }
@@ -1969,16 +1978,6 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
   RPFilter RPFilter(&F, PickedCandidateMap, TTI);
   RPFilter.run();
 
-#if 0
-  DenseMap<const BasicBlock*, unsigned> BBToNumCands;
-  for (auto &InstCand : PickedCandidateMap) {
-    const BasicBlock *BB= InstCand.first->getParent();
-    auto [It, Inserted] = BBToNumCands.try_emplace(BB);
-    if (Inserted)
-      It->second = countCandsInBB(It->first, PickedCandidateMap);
-  }
-#endif
-
   // From SortedCandidateInsts, remove some candidates that are likely to
   // increase register pressure. The candidate's Inst is the source of
   // replacement. A candidate in the following criteria should be removed:

>From 1b88d512ae9e6789b301d6f412ab3f5693614b86 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Tue, 25 Aug 2026 15:49:24 +0000
Subject: [PATCH 06/13] Add getRegisterBudget interface to TTI

---
 .../llvm/Analysis/TargetTransformInfo.h       |  11 ++
 .../llvm/Analysis/TargetTransformInfoImpl.h   |   5 +
 llvm/lib/Analysis/TargetTransformInfo.cpp     |   5 +
 .../AMDGPU/AMDGPUTargetTransformInfo.cpp      |  12 ++
 .../Target/AMDGPU/AMDGPUTargetTransformInfo.h |   1 +
 .../Scalar/StraightLineStrengthReduce.cpp     | 104 +++++++++++++++++-
 6 files changed, 132 insertions(+), 6 deletions(-)

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index e30cbc61a5420..e1d6c692f1c3b 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -1359,6 +1359,17 @@ class TargetTransformInfo {
   /// \return the number of registers in the target-provided register class.
   LLVM_ABI unsigned getNumberOfRegisters(unsigned ClassID) const;
 
+  /// \return The number of registers available to \p F before the register
+  /// allocator is forced to spill, or std::nullopt if the target cannot
+  /// provide a meaningful bound.
+  ///
+  /// Unlike getNumberOfRegisters(), which some targets deliberately
+  /// under-report to tune vectorization and interleaving, this is meant to be
+  /// a real budget usable by register-pressure heuristics. Targets with
+  /// several register files report the budget for the file that dominates
+  /// pressure.
+  LLVM_ABI std::optional<unsigned> getRegisterBudget(const Function &F) const;
+
   /// \return true if the target supports load/store that enables fault
   /// suppression of memory operands when the source condition is false.
   LLVM_ABI bool hasConditionalLoadStoreForType(Type *Ty, bool IsStore) const;
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index 625433e2e0a0b..10794868855c7 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -614,6 +614,11 @@ class LLVM_ABI TargetTransformInfoImplBase {
   }
 
   virtual unsigned getNumberOfRegisters(unsigned ClassID) const { return 8; }
+
+  virtual std::optional<unsigned> getRegisterBudget(const Function &F) const {
+    return std::nullopt;
+  }
+
   virtual bool hasConditionalLoadStoreForType(Type *Ty, bool IsStore) const {
     return false;
   }
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index 4c2cac9c440a0..a840803a096f6 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -823,6 +823,11 @@ unsigned TargetTransformInfo::getNumberOfRegisters(unsigned ClassID) const {
   return TTIImpl->getNumberOfRegisters(ClassID);
 }
 
+std::optional<unsigned>
+TargetTransformInfo::getRegisterBudget(const Function &F) const {
+  return TTIImpl->getRegisterBudget(F);
+}
+
 bool TargetTransformInfo::hasConditionalLoadStoreForType(Type *Ty,
                                                          bool IsStore) const {
   return TTIImpl->hasConditionalLoadStoreForType(Ty, IsStore);
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index ccb0c7314dcef..35a0d63e77f54 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -309,6 +309,18 @@ unsigned GCNTTIImpl::getNumberOfRegisters(unsigned RCID) const {
   return 4;
 }
 
+std::optional<unsigned> GCNTTIImpl::getRegisterBudget(const Function &F) const {
+  // Report the VGPR budget implied by the occupancy F is compiled for. Callers
+  // comparing a single lumped pressure number against this are conservative on
+  // the SGPR side, which is intentional: whether a value lands in an SGPR or a
+  // VGPR depends on divergence, not on its type.
+  //
+  // On GFX90A this is the combined VGPR+AGPR budget; see getMaxNumVectorRegs
+  // for the split. In dynamic VGPR mode "amdgpu-waves-per-eu" implies no VGPR
+  // limit at all, so this degrades to the full register file.
+  return ST->getMaxNumVGPRs(F);
+}
+
 TypeSize
 GCNTTIImpl::getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const {
   switch (K) {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
index 4d9ff8d2d767f..d327046152991 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.h
@@ -128,6 +128,7 @@ class GCNTTIImpl final : public BasicTTIImplBase<GCNTTIImpl> {
   }
 
   unsigned getNumberOfRegisters(unsigned RCID) const override;
+  std::optional<unsigned> getRegisterBudget(const Function &F) const override;
   TypeSize
   getRegisterBitWidth(TargetTransformInfo::RegisterKind Vector) const override;
   unsigned getMinVectorRegisterBitWidth() const override;
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 20a309c0091a9..6e8fb8eae1792 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -133,8 +133,31 @@ static cl::opt<int> SLSRBasisDistanceThreshold(
 static cl::opt<bool> UseTTIForRP("slsr-tti-rp", cl::init(false),
                                  cl::desc("Use TTI to compute RP for SLSR-RP"));
 
+static cl::opt<bool> EnableRPFilter(
+    "slsr-rp-filter", cl::init(false), cl::Hidden,
+    cl::desc("SLSR: skip rewrites in blocks where they would push register "
+             "pressure past the target's register budget"));
+
+static cl::opt<unsigned> SLSRRegBudget(
+    "slsr-reg-budget", cl::init(0), cl::Hidden,
+    cl::desc("SLSR: override the register budget reported by TTI"));
+
+static cl::opt<double> SLSRRPSafeFraction(
+    "slsr-rp-safe-fraction", cl::init(0.9), cl::Hidden,
+    cl::desc("SLSR: fraction of the register budget treated as safe"));
+
+static cl::opt<unsigned> SLSRRPAbsDelta(
+    "slsr-rp-abs-delta", cl::init(4), cl::Hidden,
+    cl::desc("SLSR: pressure increase tolerated in a block that is already "
+             "over the register budget"));
+
 STATISTIC(NumSCEVCandidateBasisDifferences,
           "Number of candidate-basis SCEV differences computed by SLSR");
+STATISTIC(NumRPFilteredBlocks,
+          "Number of blocks whose rewrites SLSR skipped due to register "
+          "pressure");
+STATISTIC(NumRPFilteredCandidates,
+          "Number of candidates SLSR skipped due to register pressure");
 
 namespace {
 
@@ -1558,12 +1581,21 @@ class RPFilter {
            const TargetTransformInfo *TTI)
       : F(F), PickedCandidateMap(PickedCandidateMap), TTI(TTI) {}
 
-  void run() {
+  // Candidates whose rewrite would push their block's register pressure past
+  // what the target can allocate are added to \p ToSkipRewrite.
+  void run(DenseSet<Instruction *> &ToSkipRewrite) {
     buildBBToNumCandsAndBasises(PickedCandidateMap);
     // TODO: 16 is an arbitrary threshold.
-    if (MaxNumBasisesInBB > 16)
+    bool HaveLiveness = MaxNumBasisesInBB > 16;
+    if (HaveLiveness)
       buildBBToLiveness(*F); // Do liveness analysis.
 
+    // The pressure numbers below only mean anything once liveness has been
+    // computed, and without a budget there is nothing to compare them to.
+    std::optional<unsigned> Budget;
+    if (EnableRPFilter && HaveLiveness)
+      Budget = getRegisterBudget();
+
     DEBUG_SLSR_RP(dbgs() << "-- MaxRP of BBs -- \n");
     for (auto &BB : *F) {
 #if 0
@@ -1575,10 +1607,14 @@ class RPFilter {
         dbgs() << "MaxRP:" << BB.getName() << ":" << RP << "," << RPBackward << "\n";
       });
 #else
-      auto [MaxRP, MaxRPWithSLSR] = maxPressureInBlockBackward(
-          BB, BBToLiveness[&BB].LiveIn, BBToLiveness[&BB].LiveOut);
+      const BlockLiveness &BL = getLiveness(&BB);
+      auto [MaxRP, MaxRPWithSLSR] =
+          maxPressureInBlockBackward(BB, BL.LiveIn, BL.LiveOut);
       DEBUG_SLSR_RP(dbgs() << "MaxRP:" << BB.getName() << ": (" << MaxRP << ", "
                            << MaxRPWithSLSR << ")" << "\n");
+
+      if (Budget && rewriteWouldOverflowBudget(MaxRP, MaxRPWithSLSR, *Budget))
+        skipRewritesInBlock(BB, ToSkipRewrite);
 #endif
     }
   }
@@ -1600,6 +1636,60 @@ class RPFilter {
 
   DenseMap<const BasicBlock *, BlockLiveness> BBToLiveness;
 
+  // Liveness is only computed for functions that pass the candidate-count
+  // gate. Blocks of the remaining functions read as having nothing live across
+  // their boundaries, which is why no rewrite is suppressed in that case.
+  const BlockLiveness &getLiveness(const BasicBlock *BB) const {
+    static const BlockLiveness Empty;
+    auto It = BBToLiveness.find(BB);
+    return It == BBToLiveness.end() ? Empty : It->second;
+  }
+
+  std::optional<unsigned> getRegisterBudget() const {
+    if (SLSRRegBudget > 0)
+      return SLSRRegBudget.getValue();
+    std::optional<unsigned> Budget = TTI->getRegisterBudget(*F);
+    if (Budget && *Budget == 0)
+      return std::nullopt;
+    return Budget;
+  }
+
+  // Return true if rewriting every candidate in a block, taking its peak
+  // pressure from \p Before to \p After, would ask the allocator for more
+  // registers than \p Budget.
+  bool rewriteWouldOverflowBudget(unsigned Before, unsigned After,
+                                  unsigned Budget) const {
+    // Leave the allocator some slack: it also has to satisfy register class
+    // and ABI constraints that this estimate knows nothing about.
+    unsigned SafeBudget = static_cast<unsigned>(Budget * SLSRRPSafeFraction);
+
+    // There is headroom, so however much the rewrite adds is irrelevant.
+    if (After <= SafeBudget)
+      return false;
+
+    // The rewrite is what takes the block over.
+    if (Before <= SafeBudget)
+      return true;
+
+    // Already over budget. SLSR can still lower pressure here, so only refuse
+    // rewrites that make it meaningfully worse.
+    return After > Before && After - Before > SLSRRPAbsDelta;
+  }
+
+  void skipRewritesInBlock(const BasicBlock &BB,
+                           DenseSet<Instruction *> &ToSkipRewrite) {
+    ++NumRPFilteredBlocks;
+    DEBUG_SLSR_RP(dbgs() << "RP filter: skipping rewrites in " << BB.getName()
+                         << "\n");
+    for (const Instruction &I : BB) {
+      auto It = PickedCandidateMap.find(&I);
+      if (It == PickedCandidateMap.end())
+        continue;
+      if (ToSkipRewrite.insert(It->first).second)
+        ++NumRPFilteredCandidates;
+    }
+  }
+
   std::pair<unsigned, unsigned> countCandsAndBasisesInBB(
       const BasicBlock *BB,
       const DenseMap<Instruction *, Candidate *> &PickedCandidateMap) {
@@ -1975,8 +2065,11 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
     if (Candidate *C = pickRewriteCandidate(I))
       PickedCandidateMap[I] = C;
 
+  // Candidates whose rewrite is predicted to hurt more than it helps.
+  DenseSet<Instruction *> ToSkipRewrite;
+
   RPFilter RPFilter(&F, PickedCandidateMap, TTI);
-  RPFilter.run();
+  RPFilter.run(ToSkipRewrite);
 
   // From SortedCandidateInsts, remove some candidates that are likely to
   // increase register pressure. The candidate's Inst is the source of
@@ -2004,7 +2097,6 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
   // Done before rewriting: rewriting inserts instructions and does
   // replaceAllUsesWith, which would invalidate both the in-block index map and
   // operands' user sets.
-  DenseSet<Instruction *> ToSkipRewrite;
   {
     DenseMap<const BasicBlock *, DenseMap<const Instruction *, int>> IndexCache;
     for (Instruction *I : SortedCandidateInsts)

>From 882874e7a174d56fbc1cd87892760da4da8257a6 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Tue, 1 Sep 2026 00:00:31 +0000
Subject: [PATCH 07/13] Clean up and adding a lit-test

---
 .../Scalar/StraightLineStrengthReduce.cpp     |  368 +---
 .../AMDGPU/slsr-rp-filter.ll                  | 1922 +++++++++++++++++
 .../slsr-basis-distance-threshold.ll          |   94 -
 .../slsr-rp-filter.ll                         |  296 +++
 4 files changed, 2271 insertions(+), 409 deletions(-)
 create mode 100644 llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
 delete mode 100644 llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
 create mode 100644 llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll

diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 6e8fb8eae1792..ae39823827087 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -125,16 +125,16 @@ static cl::opt<bool>
     EnablePoisonReuseGuard("enable-poison-reuse-guard", cl::init(true),
                            cl::desc("Enable poison-reuse guard"));
 
-static cl::opt<int> SLSRBasisDistanceThreshold(
-    "slsr-basis-distance-threshold", cl::init(96), cl::Hidden,
-    cl::desc("SLSR: skip rewrite if in-block distance from Basis's last "
-             "same-block use to the candidate Inst exceeds this"));
-
-static cl::opt<bool> UseTTIForRP("slsr-tti-rp", cl::init(false),
-                                 cl::desc("Use TTI to compute RP for SLSR-RP"));
+// RPFilter targets one pathological shape: a block holding many distinct
+// bases. Each basis contributes one extended live range no matter how many
+// candidates are rewritten against it, so it is the number of distinct bases,
+// not the number of candidates, that tracks how many new concurrent live
+// ranges SLSR would create. Below this count no block in the function can
+// exhibit the pathology, and the liveness and pressure analyses are skipped.
+static constexpr unsigned MinDistinctBasesToFilter = 16;
 
 static cl::opt<bool> EnableRPFilter(
-    "slsr-rp-filter", cl::init(false), cl::Hidden,
+    "slsr-rp-filter", cl::init(true), cl::Hidden,
     cl::desc("SLSR: skip rewrites in blocks where they would push register "
              "pressure past the target's register budget"));
 
@@ -626,15 +626,6 @@ class StraightLineStrengthReduce {
     if (auto *StrideInst = dyn_cast<Instruction>(C.Stride))
       PropagateDependency(StrideInst);
   };
-
-  bool hasOperandsUsedInNonRewritableUsersInAnotherBlock(
-      llvm::Instruction *Inst) const;
-  bool hasRewritableCandidates(const Instruction *Inst) const;
-  bool basisTooFarInSameBlock(
-      const Candidate &C,
-      DenseMap<const BasicBlock *, DenseMap<const Instruction *, int>>
-          &IndexCache,
-      const Instruction *Inst) const;
 };
 
 inline raw_ostream &operator<<(raw_ostream &OS,
@@ -1454,119 +1445,6 @@ bool StraightLineStrengthReduceLegacyPass::runOnFunction(Function &F) {
   return StraightLineStrengthReduce(DL, DT, SE, TTI).runOnFunction(F);
 }
 
-// Go through all operands of instruction, and check if any operand is used in
-// another block different from the instruction's block and the other use is
-// not rewritable. return true if such operand is found, otherwise return false.
-bool StraightLineStrengthReduce::
-    hasOperandsUsedInNonRewritableUsersInAnotherBlock(
-        llvm::Instruction *Inst) const {
-  llvm::BasicBlock *InstBB = Inst->getParent();
-
-  for (Value *OpVal : Inst->operand_values()) {
-    auto *OpInst = dyn_cast<Instruction>(OpVal);
-    if (!OpInst)
-      continue;
-
-    for (const User *U : OpInst->users())
-      if (auto *UI = dyn_cast<Instruction>(U)) {
-        if (UI->isDebugOrPseudoInst())
-          continue;
-        if (UI->getParent() != InstBB && !hasRewritableCandidates(UI)) {
-          LLVM_DEBUG(dbgs()
-                     << "Inst's operand is used in another block "
-                     << "("
-                     << (InstBB->hasName() ? InstBB->getName() : "unnamed")
-                     << " -> "
-                     << (UI->getParent() && UI->getParent()->hasName()
-                             ? UI->getParent()->getName()
-                             : "unnamed")
-                     << ") " << *UI << "\n");
-          return true;
-        }
-      }
-  }
-  return false;
-}
-
-bool StraightLineStrengthReduce::hasRewritableCandidates(
-    const Instruction *Inst) const {
-  if (!RewriteCandidates.count(Inst))
-    return false;
-
-  for (const Candidate *C : RewriteCandidates.at(Inst))
-    if (C->Basis)
-      return true;
-
-  return false;
-}
-
-// Assign a monotonically increasing index to each (non-debug/pseudo)
-// instruction in BB so in-block distances can be queried in O(1) once built.
-static DenseMap<const Instruction *, int>
-buildBlockIndexMap(const BasicBlock &BB) {
-  DenseMap<const Instruction *, int> IndexMap;
-  int Index = 0;
-  for (const Instruction &I : BB) {
-    // Skip debug/pseudo instructions so the distance math is identical
-    // between debug and release builds.
-    if (I.isDebugOrPseudoInst())
-      continue;
-    IndexMap[&I] = Index++;
-  }
-  return IndexMap;
-}
-
-// Return true
-// 1. if C.Basis used only in C.Ins's block before C.Ins
-// AND
-// 2. if any use of C.Basis before C.Ins and C.Ins exceeds
-// SLSRBasisDistanceThreshold.
-bool StraightLineStrengthReduce::basisTooFarInSameBlock(
-    const Candidate &C,
-    DenseMap<const BasicBlock *, DenseMap<const Instruction *, int>>
-        &IndexCache,
-    const Instruction *I) const {
-  Instruction *Inst = C.Ins;
-  assert(Inst == I);
-  Instruction *BasisInst = C.Basis ? C.Basis->Ins : nullptr;
-  if (!BasisInst)
-    return false;
-
-  const BasicBlock *BB = Inst->getParent();
-  auto [It, Inserted] = IndexCache.try_emplace(BB);
-  if (Inserted)
-    It->second = buildBlockIndexMap(*BB);
-  const DenseMap<const Instruction *, int> &IndexMap = It->second;
-
-  auto InstIt = IndexMap.find(Inst);
-  if (InstIt == IndexMap.end())
-    return false;
-  int InstIdx = InstIt->second;
-
-  int LastUseIdx = 0;
-
-  bool FoundSameBlockUse = false;
-  for (const User *U : BasisInst->users()) {
-    const auto *UI = dyn_cast<Instruction>(U);
-    if (!UI)
-      continue;
-    // If one of the uses is not in the same block, return false.
-    if (UI->getParent() != BB)
-      return false;
-    auto UseIt = IndexMap.find(UI);
-    // If any same block use is later than Inst, return false.
-    if (UseIt == IndexMap.end() || UseIt->second >= InstIdx)
-      return false;
-    FoundSameBlockUse = true;
-    LastUseIdx = std::max(LastUseIdx, UseIt->second);
-  }
-
-  if (!FoundSameBlockUse)
-    return false;
-
-  return (InstIdx - LastUseIdx) > SLSRBasisDistanceThreshold;
-}
-
 namespace {
 
 // TODO: Currently, I am considering (Basis, Cand) pair that are both in the
@@ -1581,42 +1459,53 @@ class RPFilter {
            const TargetTransformInfo *TTI)
       : F(F), PickedCandidateMap(PickedCandidateMap), TTI(TTI) {}
 
-  // Candidates whose rewrite would push their block's register pressure past
-  // what the target can allocate are added to \p ToSkipRewrite.
-  void run(DenseSet<Instruction *> &ToSkipRewrite) {
+  // Return the candidates whose rewrite would push their block's register
+  // pressure past what the target can allocate.
+  DenseSet<const Instruction *> run() {
+    DenseSet<const Instruction *> InstsToSkip;
+    if (!EnableRPFilter || PickedCandidateMap.empty())
+      return InstsToSkip;
+
+    // Without a budget there is nothing to compare the pressure against, so
+    // check for one before paying for the liveness and pressure analyses.
+    std::optional<unsigned> Budget = getRegisterBudget();
+    if (!Budget)
+      return InstsToSkip;
+
     buildBBToNumCandsAndBasises(PickedCandidateMap);
-    // TODO: 16 is an arbitrary threshold.
-    bool HaveLiveness = MaxNumBasisesInBB > 16;
-    if (HaveLiveness)
-      buildBBToLiveness(*F); // Do liveness analysis.
+    if (MaxNumBasisesInBB <= MinDistinctBasesToFilter)
+      return InstsToSkip;
 
-    // The pressure numbers below only mean anything once liveness has been
-    // computed, and without a budget there is nothing to compare them to.
-    std::optional<unsigned> Budget;
-    if (EnableRPFilter && HaveLiveness)
-      Budget = getRegisterBudget();
+    // Compute live-in and live-out of each BB in CFG
+    buildBBToLiveness(*F);
 
     DEBUG_SLSR_RP(dbgs() << "-- MaxRP of BBs -- \n");
+    SmallPtrSet<const BasicBlock *, 8> BBsToSkip;
     for (auto &BB : *F) {
-#if 0
-      unsigned RP = maxPressureInBlock(BB, BBToLiveness[&BB].LiveIn,
-                                       BBToLiveness[&BB].LiveOut);
-      unsigned RPBackward = maxPressureInBlockBackward(BB, BBToLiveness[&BB].LiveIn,
-                                      BBToLiveness[&BB].LiveOut);
-      DEBUG_WITH_TYPE("slsr-rp", {
-        dbgs() << "MaxRP:" << BB.getName() << ":" << RP << "," << RPBackward << "\n";
-      });
-#else
       const BlockLiveness &BL = getLiveness(&BB);
       auto [MaxRP, MaxRPWithSLSR] =
           maxPressureInBlockBackward(BB, BL.LiveIn, BL.LiveOut);
       DEBUG_SLSR_RP(dbgs() << "MaxRP:" << BB.getName() << ": (" << MaxRP << ", "
                            << MaxRPWithSLSR << ")" << "\n");
 
-      if (Budget && rewriteWouldOverflowBudget(MaxRP, MaxRPWithSLSR, *Budget))
-        skipRewritesInBlock(BB, ToSkipRewrite);
-#endif
-    }
+      if (!rewriteWouldOverflowBudget(MaxRP, MaxRPWithSLSR, *Budget))
+        continue;
+
+      DEBUG_SLSR_RP(dbgs() << "Skipping BB from SLSR: " << BB.getName()
+                           << "\n");
+      ++NumRPFilteredBlocks;
+      BBsToSkip.insert(&BB);
+    } // Done with BBs
+
+    // One pass over the candidates rather than over the instructions of every
+    // skipped block. A skipped block is one of the larger blocks in the
+    // function, while the candidate map is small by comparison.
+    for (const auto &It : PickedCandidateMap)
+      if (BBsToSkip.contains(It.first->getParent()) &&
+          InstsToSkip.insert(It.first).second)
+        ++NumRPFilteredCandidates;
+
+    return InstsToSkip;
   }
 
 private:
@@ -1676,20 +1565,6 @@ class RPFilter {
     return After > Before && After - Before > SLSRRPAbsDelta;
   }
 
-  void skipRewritesInBlock(const BasicBlock &BB,
-                           DenseSet<Instruction *> &ToSkipRewrite) {
-    ++NumRPFilteredBlocks;
-    DEBUG_SLSR_RP(dbgs() << "RP filter: skipping rewrites in " << BB.getName()
-                         << "\n");
-    for (const Instruction &I : BB) {
-      auto It = PickedCandidateMap.find(&I);
-      if (It == PickedCandidateMap.end())
-        continue;
-      if (ToSkipRewrite.insert(It->first).second)
-        ++NumRPFilteredCandidates;
-    }
-  }
-
   std::pair<unsigned, unsigned> countCandsAndBasisesInBB(
       const BasicBlock *BB,
       const DenseMap<Instruction *, Candidate *> &PickedCandidateMap) {
@@ -1742,28 +1617,9 @@ class RPFilter {
 
   unsigned computeRegWeight(Type *Ty) const {
     const DataLayout &DL = F->getDataLayout();
-    if (UseTTIForRP) {
-      // Aggregates legalize to a flat sequence of scalars; approximate rather
-      // than calling getRegUsageForType, which is llvm_unreachable on them.
-      if (!VectorType::isValidElementType(Ty->getScalarType()))
-        return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
-
-      // TTI's getRegUsageForType can be used for types that are
-      // validElementType(Ty->getScalarType()). However, even the valid vector
-      // type should be multiplited by get{Min}NumElements() to handle
-      // {ScalarVectorType}, FixedVectorType. Overall, getTypeSizeInBits(Ty)
-      // handing all those cases should be sufficient for heuristic.
-      unsigned RegUsage = TTI->getRegUsageForType(Ty);
-#if 0
-      if (FixedVectorType *FVT = dyn_cast<FixedVectorType>(Ty))
-        RegUsage *= FVT->getNumElements();
-      else if (ScalableVectorType *SVT = dyn_cast<ScalableVectorType>(Ty))
-        RegUsage *= SVT->getMinNumElements();
-#endif
-      return RegUsage;
-    }
 
-    // Default logic to compute RP.
+    // TTI's getRegUsageForType is less accurate than
+    // default logic to compute RP for targets like AMDGPU
     return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
   }
 
@@ -1849,79 +1705,6 @@ class RPFilter {
     return BB.getName() == "for.cond.cleanup";
   }
 
-  unsigned maxPressureInBlock(const BasicBlock &BB, const ValueSet &LiveIn,
-                              const ValueSet &LiveOut) const {
-
-#if 1
-    bool IsDebugBlock = isDebugBlock(BB);
-    unsigned StoreCount = 0;
-#endif
-
-    SmallVector<const Instruction *, 128> Order;
-    DenseMap<const Instruction *, unsigned> Idx;
-    for (const Instruction &I : BB) {
-      if (I.isDebugOrPseudoInst())
-        continue;
-      Idx[&I] = Order.size();
-      Order.push_back(&I);
-    }
-
-    DenseMap<const Value *, unsigned> LastUse;
-    for (const Instruction &I : BB) {
-      if (I.isDebugOrPseudoInst() || isa<PHINode>(&I))
-        continue;
-      for (const Value *Op : I.operand_values())
-        if (isRegisterLike(Op))
-          LastUse[Op] = Idx[&I];
-    }
-    const unsigned End = Order.size();
-    for (const Value *V : LiveOut)
-      LastUse[V] = End; // survives the block; never retires here
-
-    ValueSet Open;
-    for (const Value *V : LiveIn)
-      Open.insert(V);
-
-    unsigned MaxW = 0;
-    for (unsigned i = 0; i != End; ++i) {
-      if (isa<StoreInst>(Order[i])) {
-        StoreCount++;
-      }
-      if (!Order[i]->getType()->isVoidTy())
-        Open.insert(Order[i]);
-
-      unsigned W = 0;
-      for (const Value *V : Open) {
-        auto RW = regWeight(V);
-        W += RW;
-        if (IsDebugBlock && StoreCount == 32) {
-          DEBUG_SLSR_RP({
-            dbgs() << "regweight: " << "i: " << i << " value: " << *V
-                   << " RW: " << RW << ", ";
-          });
-          // dbgs() << "W: " << W << "\n";
-        }
-
-        // W += regWeight(V);
-      }
-      MaxW = std::max(MaxW, W);
-
-      SmallVector<const Value *, 8> Dead;
-      for (const Value *V : Open) {
-        auto It = LastUse.find(V);
-        if (It == LastUse.end() || It->second <= i)
-          Dead.push_back(V);
-      }
-      for (const Value *V : Dead)
-        Open.erase(V);
-
-      if (IsDebugBlock && StoreCount == 32) {
-        DEBUG_SLSR_RP(dbgs() << "W: " << W << " MaxW: " << MaxW << "\n");
-      }
-    }
-    return MaxW;
-  }
-
   // Return true if I is a candidate and its basis is in the same bb, false
   // otherwise.
   bool insertBasisIfCand(const Instruction *I, ValueSet &LiveSetWithSLSR,
@@ -1943,17 +1726,6 @@ class RPFilter {
     return false;
   }
 
-  // TODO: Remove
-  void updateLiveSetWithSLSR(ValueSet &LiveSetWithSLSR,
-                             const DenseSet<const Value *> &SeenLastUse,
-                             const Instruction &I, const Value *Op) const {
-    // LiveSet += {Cand.Basis} <-- Done already
-    // LiveSet -= {Op} if this is the last use of Op (i.e.
-    // SeenLastUse.contains(Op))
-    if (SeenLastUse.contains(Op))
-      LiveSetWithSLSR.erase(Op);
-  }
-
   std::pair<unsigned, unsigned>
   maxPressureInBlockBackward(const BasicBlock &BB, const ValueSet &LiveIn,
                              const ValueSet &LiveOut) const {
@@ -2047,7 +1819,7 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
   LLVM_DEBUG(dbgs() << "SLSR on Function: " << F.getName() << "\n");
   // Traverse the dominator tree in the depth-first order. This order makes sure
   // all bases of a candidate are in Candidates when we process it.
-  for (auto *const Node : depth_first(DT))
+  for (const auto Node : depth_first(DT))
     for (auto &I : *(Node->getBlock()))
       allocateCandidatesAndFindBasis(&I);
 
@@ -2059,52 +1831,18 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
   }
   sortCandidateInstructions();
 
-  ////////////////////////////////////////////////////
   DenseMap<Instruction *, Candidate *> PickedCandidateMap;
   for (Instruction *I : SortedCandidateInsts)
     if (Candidate *C = pickRewriteCandidate(I))
       PickedCandidateMap[I] = C;
 
-  // Candidates whose rewrite is predicted to hurt more than it helps.
-  DenseSet<Instruction *> ToSkipRewrite;
-
+  // Candidates whose rewrite would push their block's register pressure past
+  // what the target can allocate. Evaluated on the original IR, before any
+  // rewriteCandidate mutates it: rewriting inserts instructions and calls
+  // replaceAllUsesWith, which would invalidate the liveness and pressure
+  // analyses the filter relies on.
   RPFilter RPFilter(&F, PickedCandidateMap, TTI);
-  RPFilter.run(ToSkipRewrite);
-
-  // From SortedCandidateInsts, remove some candidates that are likely to
-  // increase register pressure. The candidate's Inst is the source of
-  // replacement. A candidate in the following criteria should be removed:
-  // 1. The candidate's Inst "has operands used in non-rewritable users in
-  // another block"
-  //    -- checked by hasOperandsUsedInNonRewritableUsersInAnotherBlock(Inst)
-  //    -- This means the candidate's Inst's original operands are live in
-  //    another block, so even if rewrite the Inst, the operands may be still
-  //    live out to another block.
-  //    -- Thus, rewriting the Inst based on Basis might add another long live
-  //    range from the Basis by increasing the live range of the Basis.
-  //    -- TODO: If needed, a refinement to check that "another block" is
-  //    properly dominated by the candidate's Inst's block can be added.
-  // 2. When the candidate's Basis's is only used in the the same block
-  // and its last use is before the candidate's Inst, the difference between the
-  // last use of Basis and the Inst is larger than a threshold.
-  //    -- This is also for avoiding increasing the live range of the Basis by
-  //    rewriting the Inst.
-  //
-  // A candidate satisfies both conditions 1 and 2 should be removed.
-
-  // Collect candidates likely to increase register pressure.
-  // Evaluate on the original IR, before any rewriteCandidate mutates it
-  // Done before rewriting: rewriting inserts instructions and does
-  // replaceAllUsesWith, which would invalidate both the in-block index map and
-  // operands' user sets.
-  {
-    DenseMap<const BasicBlock *, DenseMap<const Instruction *, int>> IndexCache;
-    for (Instruction *I : SortedCandidateInsts)
-      if (Candidate *C = pickRewriteCandidate(I))
-        if (hasOperandsUsedInNonRewritableUsersInAnotherBlock(I) &&
-            basisTooFarInSameBlock(*C, IndexCache, I))
-          ToSkipRewrite.insert(I);
-  }
+  DenseSet<const Instruction *> ToSkipRewrite = RPFilter.run();
 
   // Rewrite candidates in the topological order that rewrites a Candidate
   // always before rewriting its Basis
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
new file mode 100644
index 0000000000000..24ae488d9a7f3
--- /dev/null
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
@@ -0,0 +1,1922 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -mtriple=amdgpu9.50-amd-amdhsa -passes=slsr -S -slsr-reg-budget=32 | FileCheck %s --check-prefixes=CHECK,BUDGET32
+; RUN: opt < %s -mtriple=amdgpu9.50-amd-amdhsa -passes=slsr -S | FileCheck %s --check-prefixes=CHECK,ATTRS
+; RUN: opt < %s -mtriple=amdgpu9.50-amd-amdhsa -passes=slsr -S -slsr-rp-filter=false | FileCheck %s --check-prefixes=CHECK,NOFILTER
+
+; SLSR rewrites a candidate as its basis plus a delta, which keeps the basis live
+; down to the candidate. The register-pressure filter drops a block's rewrites
+; when doing so would take the block's peak pressure past what the target can
+; allocate.
+;
+; The budget comes from TTI, which on AMDGPU reports the VGPR count implied by
+; the occupancy the function is compiled for, and -slsr-reg-budget overrides it.
+; -slsr-rp-safe-fraction (0.9 by default) is applied on top, so a budget of 32
+; leaves 28 usable registers. The filter does nothing unless -slsr-rp-filter is
+; passed, and it only examines functions where some block holds more than 16
+; distinct bases.
+;
+; @many_bases_overlapping has 17 bases and goes from 20 to 36 registers, crossing
+; the 28-register safe budget, so its rewrites are dropped. @many_bases_adjacent
+; has the same 17 bases, but each %u<k> immediately follows its %t<k>, so the
+; extended ranges never coexist and the peak only reaches 21, which fits.
+;
+; @four_bases holds only 4 bases, below the gate, so the filter never looks at it
+; and its rewrites stand. Its i128 values put the block at 28 registers rising to
+; 44, which would cross the 28-register safe budget if it were examined, so the
+; gate really is the only thing sparing it.
+;
+; The i32 functions carry no occupancy attributes, so their flat work group size
+; defaults to 1024, which is 16 wave64s spread over 4 EUs, hence 4 waves per EU
+; and a budget of 512/4 = 128 registers. Their peak of 36 fits, so the ATTRS run
+; leaves them alone and only the forced budget of 32 filters them.
+;
+; The i128 functions at the end raise the peak to 144 so the real budget decides
+; without any override. All three have identical bodies and identical pressure,
+; so the attributes are the only variable:
+;
+;   @default_occupancy      no attributes           4 waves/EU  budget 128  filtered
+;   @small_work_group       flat-work-group-size    1 wave/EU   budget 512  kept
+;   @small_wg_high_occ      + waves-per-eu=8        8 waves/EU  budget 64   filtered
+;
+; The first pair shows flat-work-group-size raising the budget, and the second
+; shows waves-per-eu lowering it again. NOFILTER switches the analysis off.
+
+declare void @foo(i32)
+declare void @bar(i128)
+
+define void @many_bases_overlapping(i32 %s, i32 %b1, i32 %b2, i32 %b3, i32 %b4, i32 %b5, i32 %b6, i32 %b7, i32 %b8, i32 %b9, i32 %b10, i32 %b11, i32 %b12, i32 %b13, i32 %b14, i32 %b15, i32 %b16, i32 %b17) {
+; FILTER-LABEL: define void @many_bases_overlapping(
+; FILTER-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; FILTER-NEXT:  [[ENTRY:.*:]]
+; FILTER-NEXT:    [[S2:%.*]] = shl i32 [[S]], 1
+; FILTER-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T1]])
+; FILTER-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T2]])
+; FILTER-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T3]])
+; FILTER-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T4]])
+; FILTER-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T5]])
+; FILTER-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T6]])
+; FILTER-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T7]])
+; FILTER-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T8]])
+; FILTER-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T9]])
+; FILTER-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T10]])
+; FILTER-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T11]])
+; FILTER-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T12]])
+; FILTER-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T13]])
+; FILTER-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T14]])
+; FILTER-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T15]])
+; FILTER-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T16]])
+; FILTER-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
+; FILTER-NEXT:    call void @foo(i32 [[T17]])
+; FILTER-NEXT:    [[U1:%.*]] = add i32 [[B1]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U1]])
+; FILTER-NEXT:    [[U2:%.*]] = add i32 [[B2]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U2]])
+; FILTER-NEXT:    [[U3:%.*]] = add i32 [[B3]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U3]])
+; FILTER-NEXT:    [[U4:%.*]] = add i32 [[B4]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U4]])
+; FILTER-NEXT:    [[U5:%.*]] = add i32 [[B5]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U5]])
+; FILTER-NEXT:    [[U6:%.*]] = add i32 [[B6]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U6]])
+; FILTER-NEXT:    [[U7:%.*]] = add i32 [[B7]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U7]])
+; FILTER-NEXT:    [[U8:%.*]] = add i32 [[B8]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U8]])
+; FILTER-NEXT:    [[U9:%.*]] = add i32 [[B9]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U9]])
+; FILTER-NEXT:    [[U10:%.*]] = add i32 [[B10]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U10]])
+; FILTER-NEXT:    [[U11:%.*]] = add i32 [[B11]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U11]])
+; FILTER-NEXT:    [[U12:%.*]] = add i32 [[B12]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U12]])
+; FILTER-NEXT:    [[U13:%.*]] = add i32 [[B13]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U13]])
+; FILTER-NEXT:    [[U14:%.*]] = add i32 [[B14]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U14]])
+; FILTER-NEXT:    [[U15:%.*]] = add i32 [[B15]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U15]])
+; FILTER-NEXT:    [[U16:%.*]] = add i32 [[B16]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U16]])
+; FILTER-NEXT:    [[U17:%.*]] = add i32 [[B17]], [[S2]]
+; FILTER-NEXT:    call void @foo(i32 [[U17]])
+; FILTER-NEXT:    call void @foo(i32 [[B1]])
+; FILTER-NEXT:    call void @foo(i32 [[B2]])
+; FILTER-NEXT:    call void @foo(i32 [[B3]])
+; FILTER-NEXT:    call void @foo(i32 [[B4]])
+; FILTER-NEXT:    call void @foo(i32 [[B5]])
+; FILTER-NEXT:    call void @foo(i32 [[B6]])
+; FILTER-NEXT:    call void @foo(i32 [[B7]])
+; FILTER-NEXT:    call void @foo(i32 [[B8]])
+; FILTER-NEXT:    call void @foo(i32 [[B9]])
+; FILTER-NEXT:    call void @foo(i32 [[B10]])
+; FILTER-NEXT:    call void @foo(i32 [[B11]])
+; FILTER-NEXT:    call void @foo(i32 [[B12]])
+; FILTER-NEXT:    call void @foo(i32 [[B13]])
+; FILTER-NEXT:    call void @foo(i32 [[B14]])
+; FILTER-NEXT:    call void @foo(i32 [[B15]])
+; FILTER-NEXT:    call void @foo(i32 [[B16]])
+; FILTER-NEXT:    call void @foo(i32 [[B17]])
+; FILTER-NEXT:    ret void
+;
+; OFF-LABEL: define void @many_bases_overlapping(
+; OFF-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; OFF-NEXT:  [[ENTRY:.*:]]
+; OFF-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T1]])
+; OFF-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T2]])
+; OFF-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T3]])
+; OFF-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T4]])
+; OFF-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T5]])
+; OFF-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T6]])
+; OFF-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T7]])
+; OFF-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T8]])
+; OFF-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T9]])
+; OFF-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T10]])
+; OFF-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T11]])
+; OFF-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T12]])
+; OFF-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T13]])
+; OFF-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T14]])
+; OFF-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T15]])
+; OFF-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T16]])
+; OFF-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[T17]])
+; OFF-NEXT:    [[U1:%.*]] = add i32 [[T1]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U1]])
+; OFF-NEXT:    [[U2:%.*]] = add i32 [[T2]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U2]])
+; OFF-NEXT:    [[U3:%.*]] = add i32 [[T3]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U3]])
+; OFF-NEXT:    [[U4:%.*]] = add i32 [[T4]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U4]])
+; OFF-NEXT:    [[U5:%.*]] = add i32 [[T5]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U5]])
+; OFF-NEXT:    [[U6:%.*]] = add i32 [[T6]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U6]])
+; OFF-NEXT:    [[U7:%.*]] = add i32 [[T7]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U7]])
+; OFF-NEXT:    [[U8:%.*]] = add i32 [[T8]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U8]])
+; OFF-NEXT:    [[U9:%.*]] = add i32 [[T9]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U9]])
+; OFF-NEXT:    [[U10:%.*]] = add i32 [[T10]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U10]])
+; OFF-NEXT:    [[U11:%.*]] = add i32 [[T11]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U11]])
+; OFF-NEXT:    [[U12:%.*]] = add i32 [[T12]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U12]])
+; OFF-NEXT:    [[U13:%.*]] = add i32 [[T13]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U13]])
+; OFF-NEXT:    [[U14:%.*]] = add i32 [[T14]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U14]])
+; OFF-NEXT:    [[U15:%.*]] = add i32 [[T15]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U15]])
+; OFF-NEXT:    [[U16:%.*]] = add i32 [[T16]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U16]])
+; OFF-NEXT:    [[U17:%.*]] = add i32 [[T17]], [[S]]
+; OFF-NEXT:    call void @foo(i32 [[U17]])
+; OFF-NEXT:    call void @foo(i32 [[B1]])
+; OFF-NEXT:    call void @foo(i32 [[B2]])
+; OFF-NEXT:    call void @foo(i32 [[B3]])
+; OFF-NEXT:    call void @foo(i32 [[B4]])
+; OFF-NEXT:    call void @foo(i32 [[B5]])
+; OFF-NEXT:    call void @foo(i32 [[B6]])
+; OFF-NEXT:    call void @foo(i32 [[B7]])
+; OFF-NEXT:    call void @foo(i32 [[B8]])
+; OFF-NEXT:    call void @foo(i32 [[B9]])
+; OFF-NEXT:    call void @foo(i32 [[B10]])
+; OFF-NEXT:    call void @foo(i32 [[B11]])
+; OFF-NEXT:    call void @foo(i32 [[B12]])
+; OFF-NEXT:    call void @foo(i32 [[B13]])
+; OFF-NEXT:    call void @foo(i32 [[B14]])
+; OFF-NEXT:    call void @foo(i32 [[B15]])
+; OFF-NEXT:    call void @foo(i32 [[B16]])
+; OFF-NEXT:    call void @foo(i32 [[B17]])
+; OFF-NEXT:    ret void
+;
+; BUDGET32-LABEL: define void @many_bases_overlapping(
+; BUDGET32-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; BUDGET32-NEXT:  [[ENTRY:.*:]]
+; BUDGET32-NEXT:    [[S2:%.*]] = shl i32 [[S]], 1
+; BUDGET32-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T1]])
+; BUDGET32-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T2]])
+; BUDGET32-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T3]])
+; BUDGET32-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T4]])
+; BUDGET32-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T5]])
+; BUDGET32-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T6]])
+; BUDGET32-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T7]])
+; BUDGET32-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T8]])
+; BUDGET32-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T9]])
+; BUDGET32-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T10]])
+; BUDGET32-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T11]])
+; BUDGET32-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T12]])
+; BUDGET32-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T13]])
+; BUDGET32-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T14]])
+; BUDGET32-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T15]])
+; BUDGET32-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T16]])
+; BUDGET32-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
+; BUDGET32-NEXT:    call void @foo(i32 [[T17]])
+; BUDGET32-NEXT:    [[U1:%.*]] = add i32 [[B1]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U1]])
+; BUDGET32-NEXT:    [[U2:%.*]] = add i32 [[B2]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U2]])
+; BUDGET32-NEXT:    [[U3:%.*]] = add i32 [[B3]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U3]])
+; BUDGET32-NEXT:    [[U4:%.*]] = add i32 [[B4]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U4]])
+; BUDGET32-NEXT:    [[U5:%.*]] = add i32 [[B5]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U5]])
+; BUDGET32-NEXT:    [[U6:%.*]] = add i32 [[B6]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U6]])
+; BUDGET32-NEXT:    [[U7:%.*]] = add i32 [[B7]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U7]])
+; BUDGET32-NEXT:    [[U8:%.*]] = add i32 [[B8]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U8]])
+; BUDGET32-NEXT:    [[U9:%.*]] = add i32 [[B9]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U9]])
+; BUDGET32-NEXT:    [[U10:%.*]] = add i32 [[B10]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U10]])
+; BUDGET32-NEXT:    [[U11:%.*]] = add i32 [[B11]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U11]])
+; BUDGET32-NEXT:    [[U12:%.*]] = add i32 [[B12]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U12]])
+; BUDGET32-NEXT:    [[U13:%.*]] = add i32 [[B13]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U13]])
+; BUDGET32-NEXT:    [[U14:%.*]] = add i32 [[B14]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U14]])
+; BUDGET32-NEXT:    [[U15:%.*]] = add i32 [[B15]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U15]])
+; BUDGET32-NEXT:    [[U16:%.*]] = add i32 [[B16]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U16]])
+; BUDGET32-NEXT:    [[U17:%.*]] = add i32 [[B17]], [[S2]]
+; BUDGET32-NEXT:    call void @foo(i32 [[U17]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B1]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B2]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B3]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B4]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B5]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B6]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B7]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B8]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B9]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B10]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B11]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B12]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B13]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B14]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B15]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B16]])
+; BUDGET32-NEXT:    call void @foo(i32 [[B17]])
+; BUDGET32-NEXT:    ret void
+;
+; ATTRS-LABEL: define void @many_bases_overlapping(
+; ATTRS-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; ATTRS-NEXT:  [[ENTRY:.*:]]
+; ATTRS-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T1]])
+; ATTRS-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T2]])
+; ATTRS-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T3]])
+; ATTRS-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T4]])
+; ATTRS-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T5]])
+; ATTRS-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T6]])
+; ATTRS-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T7]])
+; ATTRS-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T8]])
+; ATTRS-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T9]])
+; ATTRS-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T10]])
+; ATTRS-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T11]])
+; ATTRS-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T12]])
+; ATTRS-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T13]])
+; ATTRS-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T14]])
+; ATTRS-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T15]])
+; ATTRS-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T16]])
+; ATTRS-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[T17]])
+; ATTRS-NEXT:    [[U1:%.*]] = add i32 [[T1]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U1]])
+; ATTRS-NEXT:    [[U2:%.*]] = add i32 [[T2]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U2]])
+; ATTRS-NEXT:    [[U3:%.*]] = add i32 [[T3]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U3]])
+; ATTRS-NEXT:    [[U4:%.*]] = add i32 [[T4]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U4]])
+; ATTRS-NEXT:    [[U5:%.*]] = add i32 [[T5]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U5]])
+; ATTRS-NEXT:    [[U6:%.*]] = add i32 [[T6]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U6]])
+; ATTRS-NEXT:    [[U7:%.*]] = add i32 [[T7]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U7]])
+; ATTRS-NEXT:    [[U8:%.*]] = add i32 [[T8]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U8]])
+; ATTRS-NEXT:    [[U9:%.*]] = add i32 [[T9]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U9]])
+; ATTRS-NEXT:    [[U10:%.*]] = add i32 [[T10]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U10]])
+; ATTRS-NEXT:    [[U11:%.*]] = add i32 [[T11]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U11]])
+; ATTRS-NEXT:    [[U12:%.*]] = add i32 [[T12]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U12]])
+; ATTRS-NEXT:    [[U13:%.*]] = add i32 [[T13]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U13]])
+; ATTRS-NEXT:    [[U14:%.*]] = add i32 [[T14]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U14]])
+; ATTRS-NEXT:    [[U15:%.*]] = add i32 [[T15]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U15]])
+; ATTRS-NEXT:    [[U16:%.*]] = add i32 [[T16]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U16]])
+; ATTRS-NEXT:    [[U17:%.*]] = add i32 [[T17]], [[S]]
+; ATTRS-NEXT:    call void @foo(i32 [[U17]])
+; ATTRS-NEXT:    call void @foo(i32 [[B1]])
+; ATTRS-NEXT:    call void @foo(i32 [[B2]])
+; ATTRS-NEXT:    call void @foo(i32 [[B3]])
+; ATTRS-NEXT:    call void @foo(i32 [[B4]])
+; ATTRS-NEXT:    call void @foo(i32 [[B5]])
+; ATTRS-NEXT:    call void @foo(i32 [[B6]])
+; ATTRS-NEXT:    call void @foo(i32 [[B7]])
+; ATTRS-NEXT:    call void @foo(i32 [[B8]])
+; ATTRS-NEXT:    call void @foo(i32 [[B9]])
+; ATTRS-NEXT:    call void @foo(i32 [[B10]])
+; ATTRS-NEXT:    call void @foo(i32 [[B11]])
+; ATTRS-NEXT:    call void @foo(i32 [[B12]])
+; ATTRS-NEXT:    call void @foo(i32 [[B13]])
+; ATTRS-NEXT:    call void @foo(i32 [[B14]])
+; ATTRS-NEXT:    call void @foo(i32 [[B15]])
+; ATTRS-NEXT:    call void @foo(i32 [[B16]])
+; ATTRS-NEXT:    call void @foo(i32 [[B17]])
+; ATTRS-NEXT:    ret void
+;
+; NOFILTER-LABEL: define void @many_bases_overlapping(
+; NOFILTER-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; NOFILTER-NEXT:  [[ENTRY:.*:]]
+; NOFILTER-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T1]])
+; NOFILTER-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T2]])
+; NOFILTER-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T3]])
+; NOFILTER-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T4]])
+; NOFILTER-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T5]])
+; NOFILTER-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T6]])
+; NOFILTER-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T7]])
+; NOFILTER-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T8]])
+; NOFILTER-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T9]])
+; NOFILTER-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T10]])
+; NOFILTER-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T11]])
+; NOFILTER-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T12]])
+; NOFILTER-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T13]])
+; NOFILTER-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T14]])
+; NOFILTER-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T15]])
+; NOFILTER-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T16]])
+; NOFILTER-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[T17]])
+; NOFILTER-NEXT:    [[U1:%.*]] = add i32 [[T1]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U1]])
+; NOFILTER-NEXT:    [[U2:%.*]] = add i32 [[T2]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U2]])
+; NOFILTER-NEXT:    [[U3:%.*]] = add i32 [[T3]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U3]])
+; NOFILTER-NEXT:    [[U4:%.*]] = add i32 [[T4]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U4]])
+; NOFILTER-NEXT:    [[U5:%.*]] = add i32 [[T5]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U5]])
+; NOFILTER-NEXT:    [[U6:%.*]] = add i32 [[T6]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U6]])
+; NOFILTER-NEXT:    [[U7:%.*]] = add i32 [[T7]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U7]])
+; NOFILTER-NEXT:    [[U8:%.*]] = add i32 [[T8]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U8]])
+; NOFILTER-NEXT:    [[U9:%.*]] = add i32 [[T9]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U9]])
+; NOFILTER-NEXT:    [[U10:%.*]] = add i32 [[T10]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U10]])
+; NOFILTER-NEXT:    [[U11:%.*]] = add i32 [[T11]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U11]])
+; NOFILTER-NEXT:    [[U12:%.*]] = add i32 [[T12]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U12]])
+; NOFILTER-NEXT:    [[U13:%.*]] = add i32 [[T13]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U13]])
+; NOFILTER-NEXT:    [[U14:%.*]] = add i32 [[T14]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U14]])
+; NOFILTER-NEXT:    [[U15:%.*]] = add i32 [[T15]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U15]])
+; NOFILTER-NEXT:    [[U16:%.*]] = add i32 [[T16]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U16]])
+; NOFILTER-NEXT:    [[U17:%.*]] = add i32 [[T17]], [[S]]
+; NOFILTER-NEXT:    call void @foo(i32 [[U17]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B1]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B2]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B3]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B4]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B5]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B6]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B7]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B8]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B9]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B10]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B11]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B12]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B13]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B14]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B15]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B16]])
+; NOFILTER-NEXT:    call void @foo(i32 [[B17]])
+; NOFILTER-NEXT:    ret void
+;
+entry:
+  %s2 = shl i32 %s, 1
+  %t1 = add i32 %b1, %s
+  call void @foo(i32 %t1)
+  %t2 = add i32 %b2, %s
+  call void @foo(i32 %t2)
+  %t3 = add i32 %b3, %s
+  call void @foo(i32 %t3)
+  %t4 = add i32 %b4, %s
+  call void @foo(i32 %t4)
+  %t5 = add i32 %b5, %s
+  call void @foo(i32 %t5)
+  %t6 = add i32 %b6, %s
+  call void @foo(i32 %t6)
+  %t7 = add i32 %b7, %s
+  call void @foo(i32 %t7)
+  %t8 = add i32 %b8, %s
+  call void @foo(i32 %t8)
+  %t9 = add i32 %b9, %s
+  call void @foo(i32 %t9)
+  %t10 = add i32 %b10, %s
+  call void @foo(i32 %t10)
+  %t11 = add i32 %b11, %s
+  call void @foo(i32 %t11)
+  %t12 = add i32 %b12, %s
+  call void @foo(i32 %t12)
+  %t13 = add i32 %b13, %s
+  call void @foo(i32 %t13)
+  %t14 = add i32 %b14, %s
+  call void @foo(i32 %t14)
+  %t15 = add i32 %b15, %s
+  call void @foo(i32 %t15)
+  %t16 = add i32 %b16, %s
+  call void @foo(i32 %t16)
+  %t17 = add i32 %b17, %s
+  call void @foo(i32 %t17)
+  %u1 = add i32 %b1, %s2
+  call void @foo(i32 %u1)
+  %u2 = add i32 %b2, %s2
+  call void @foo(i32 %u2)
+  %u3 = add i32 %b3, %s2
+  call void @foo(i32 %u3)
+  %u4 = add i32 %b4, %s2
+  call void @foo(i32 %u4)
+  %u5 = add i32 %b5, %s2
+  call void @foo(i32 %u5)
+  %u6 = add i32 %b6, %s2
+  call void @foo(i32 %u6)
+  %u7 = add i32 %b7, %s2
+  call void @foo(i32 %u7)
+  %u8 = add i32 %b8, %s2
+  call void @foo(i32 %u8)
+  %u9 = add i32 %b9, %s2
+  call void @foo(i32 %u9)
+  %u10 = add i32 %b10, %s2
+  call void @foo(i32 %u10)
+  %u11 = add i32 %b11, %s2
+  call void @foo(i32 %u11)
+  %u12 = add i32 %b12, %s2
+  call void @foo(i32 %u12)
+  %u13 = add i32 %b13, %s2
+  call void @foo(i32 %u13)
+  %u14 = add i32 %b14, %s2
+  call void @foo(i32 %u14)
+  %u15 = add i32 %b15, %s2
+  call void @foo(i32 %u15)
+  %u16 = add i32 %b16, %s2
+  call void @foo(i32 %u16)
+  %u17 = add i32 %b17, %s2
+  call void @foo(i32 %u17)
+  call void @foo(i32 %b1)
+  call void @foo(i32 %b2)
+  call void @foo(i32 %b3)
+  call void @foo(i32 %b4)
+  call void @foo(i32 %b5)
+  call void @foo(i32 %b6)
+  call void @foo(i32 %b7)
+  call void @foo(i32 %b8)
+  call void @foo(i32 %b9)
+  call void @foo(i32 %b10)
+  call void @foo(i32 %b11)
+  call void @foo(i32 %b12)
+  call void @foo(i32 %b13)
+  call void @foo(i32 %b14)
+  call void @foo(i32 %b15)
+  call void @foo(i32 %b16)
+  call void @foo(i32 %b17)
+  ret void
+}
+
+define void @many_bases_adjacent(i32 %s, i32 %b1, i32 %b2, i32 %b3, i32 %b4, i32 %b5, i32 %b6, i32 %b7, i32 %b8, i32 %b9, i32 %b10, i32 %b11, i32 %b12, i32 %b13, i32 %b14, i32 %b15, i32 %b16, i32 %b17) {
+; CHECK-LABEL: define void @many_bases_adjacent(
+; CHECK-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T1]])
+; CHECK-NEXT:    [[U1:%.*]] = add i32 [[T1]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U1]])
+; CHECK-NEXT:    call void @foo(i32 [[B1]])
+; CHECK-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T2]])
+; CHECK-NEXT:    [[U2:%.*]] = add i32 [[T2]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U2]])
+; CHECK-NEXT:    call void @foo(i32 [[B2]])
+; CHECK-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T3]])
+; CHECK-NEXT:    [[U3:%.*]] = add i32 [[T3]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U3]])
+; CHECK-NEXT:    call void @foo(i32 [[B3]])
+; CHECK-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T4]])
+; CHECK-NEXT:    [[U4:%.*]] = add i32 [[T4]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U4]])
+; CHECK-NEXT:    call void @foo(i32 [[B4]])
+; CHECK-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T5]])
+; CHECK-NEXT:    [[U5:%.*]] = add i32 [[T5]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U5]])
+; CHECK-NEXT:    call void @foo(i32 [[B5]])
+; CHECK-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T6]])
+; CHECK-NEXT:    [[U6:%.*]] = add i32 [[T6]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U6]])
+; CHECK-NEXT:    call void @foo(i32 [[B6]])
+; CHECK-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T7]])
+; CHECK-NEXT:    [[U7:%.*]] = add i32 [[T7]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U7]])
+; CHECK-NEXT:    call void @foo(i32 [[B7]])
+; CHECK-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T8]])
+; CHECK-NEXT:    [[U8:%.*]] = add i32 [[T8]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U8]])
+; CHECK-NEXT:    call void @foo(i32 [[B8]])
+; CHECK-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T9]])
+; CHECK-NEXT:    [[U9:%.*]] = add i32 [[T9]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U9]])
+; CHECK-NEXT:    call void @foo(i32 [[B9]])
+; CHECK-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T10]])
+; CHECK-NEXT:    [[U10:%.*]] = add i32 [[T10]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U10]])
+; CHECK-NEXT:    call void @foo(i32 [[B10]])
+; CHECK-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T11]])
+; CHECK-NEXT:    [[U11:%.*]] = add i32 [[T11]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U11]])
+; CHECK-NEXT:    call void @foo(i32 [[B11]])
+; CHECK-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T12]])
+; CHECK-NEXT:    [[U12:%.*]] = add i32 [[T12]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U12]])
+; CHECK-NEXT:    call void @foo(i32 [[B12]])
+; CHECK-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T13]])
+; CHECK-NEXT:    [[U13:%.*]] = add i32 [[T13]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U13]])
+; CHECK-NEXT:    call void @foo(i32 [[B13]])
+; CHECK-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T14]])
+; CHECK-NEXT:    [[U14:%.*]] = add i32 [[T14]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U14]])
+; CHECK-NEXT:    call void @foo(i32 [[B14]])
+; CHECK-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T15]])
+; CHECK-NEXT:    [[U15:%.*]] = add i32 [[T15]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U15]])
+; CHECK-NEXT:    call void @foo(i32 [[B15]])
+; CHECK-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T16]])
+; CHECK-NEXT:    [[U16:%.*]] = add i32 [[T16]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U16]])
+; CHECK-NEXT:    call void @foo(i32 [[B16]])
+; CHECK-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[T17]])
+; CHECK-NEXT:    [[U17:%.*]] = add i32 [[T17]], [[S]]
+; CHECK-NEXT:    call void @foo(i32 [[U17]])
+; CHECK-NEXT:    call void @foo(i32 [[B17]])
+; CHECK-NEXT:    ret void
+;
+entry:
+  %s2 = shl i32 %s, 1
+  %t1 = add i32 %b1, %s
+  call void @foo(i32 %t1)
+  %u1 = add i32 %b1, %s2
+  call void @foo(i32 %u1)
+  call void @foo(i32 %b1)
+  %t2 = add i32 %b2, %s
+  call void @foo(i32 %t2)
+  %u2 = add i32 %b2, %s2
+  call void @foo(i32 %u2)
+  call void @foo(i32 %b2)
+  %t3 = add i32 %b3, %s
+  call void @foo(i32 %t3)
+  %u3 = add i32 %b3, %s2
+  call void @foo(i32 %u3)
+  call void @foo(i32 %b3)
+  %t4 = add i32 %b4, %s
+  call void @foo(i32 %t4)
+  %u4 = add i32 %b4, %s2
+  call void @foo(i32 %u4)
+  call void @foo(i32 %b4)
+  %t5 = add i32 %b5, %s
+  call void @foo(i32 %t5)
+  %u5 = add i32 %b5, %s2
+  call void @foo(i32 %u5)
+  call void @foo(i32 %b5)
+  %t6 = add i32 %b6, %s
+  call void @foo(i32 %t6)
+  %u6 = add i32 %b6, %s2
+  call void @foo(i32 %u6)
+  call void @foo(i32 %b6)
+  %t7 = add i32 %b7, %s
+  call void @foo(i32 %t7)
+  %u7 = add i32 %b7, %s2
+  call void @foo(i32 %u7)
+  call void @foo(i32 %b7)
+  %t8 = add i32 %b8, %s
+  call void @foo(i32 %t8)
+  %u8 = add i32 %b8, %s2
+  call void @foo(i32 %u8)
+  call void @foo(i32 %b8)
+  %t9 = add i32 %b9, %s
+  call void @foo(i32 %t9)
+  %u9 = add i32 %b9, %s2
+  call void @foo(i32 %u9)
+  call void @foo(i32 %b9)
+  %t10 = add i32 %b10, %s
+  call void @foo(i32 %t10)
+  %u10 = add i32 %b10, %s2
+  call void @foo(i32 %u10)
+  call void @foo(i32 %b10)
+  %t11 = add i32 %b11, %s
+  call void @foo(i32 %t11)
+  %u11 = add i32 %b11, %s2
+  call void @foo(i32 %u11)
+  call void @foo(i32 %b11)
+  %t12 = add i32 %b12, %s
+  call void @foo(i32 %t12)
+  %u12 = add i32 %b12, %s2
+  call void @foo(i32 %u12)
+  call void @foo(i32 %b12)
+  %t13 = add i32 %b13, %s
+  call void @foo(i32 %t13)
+  %u13 = add i32 %b13, %s2
+  call void @foo(i32 %u13)
+  call void @foo(i32 %b13)
+  %t14 = add i32 %b14, %s
+  call void @foo(i32 %t14)
+  %u14 = add i32 %b14, %s2
+  call void @foo(i32 %u14)
+  call void @foo(i32 %b14)
+  %t15 = add i32 %b15, %s
+  call void @foo(i32 %t15)
+  %u15 = add i32 %b15, %s2
+  call void @foo(i32 %u15)
+  call void @foo(i32 %b15)
+  %t16 = add i32 %b16, %s
+  call void @foo(i32 %t16)
+  %u16 = add i32 %b16, %s2
+  call void @foo(i32 %u16)
+  call void @foo(i32 %b16)
+  %t17 = add i32 %b17, %s
+  call void @foo(i32 %t17)
+  %u17 = add i32 %b17, %s2
+  call void @foo(i32 %u17)
+  call void @foo(i32 %b17)
+  ret void
+}
+
+define void @four_bases(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4) {
+; CHECK-LABEL: define void @four_bases(
+; CHECK-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T1]])
+; CHECK-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T2]])
+; CHECK-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T3]])
+; CHECK-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T4]])
+; CHECK-NEXT:    [[U1:%.*]] = add i128 [[T1]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U1]])
+; CHECK-NEXT:    [[U2:%.*]] = add i128 [[T2]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U2]])
+; CHECK-NEXT:    [[U3:%.*]] = add i128 [[T3]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U3]])
+; CHECK-NEXT:    [[U4:%.*]] = add i128 [[T4]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U4]])
+; CHECK-NEXT:    call void @bar(i128 [[B1]])
+; CHECK-NEXT:    call void @bar(i128 [[B2]])
+; CHECK-NEXT:    call void @bar(i128 [[B3]])
+; CHECK-NEXT:    call void @bar(i128 [[B4]])
+; CHECK-NEXT:    ret void
+;
+entry:
+  %s2 = shl i128 %s, 1
+  %t1 = add i128 %b1, %s
+  call void @bar(i128 %t1)
+  %t2 = add i128 %b2, %s
+  call void @bar(i128 %t2)
+  %t3 = add i128 %b3, %s
+  call void @bar(i128 %t3)
+  %t4 = add i128 %b4, %s
+  call void @bar(i128 %t4)
+  %u1 = add i128 %b1, %s2
+  call void @bar(i128 %u1)
+  %u2 = add i128 %b2, %s2
+  call void @bar(i128 %u2)
+  %u3 = add i128 %b3, %s2
+  call void @bar(i128 %u3)
+  %u4 = add i128 %b4, %s2
+  call void @bar(i128 %u4)
+  call void @bar(i128 %b1)
+  call void @bar(i128 %b2)
+  call void @bar(i128 %b3)
+  call void @bar(i128 %b4)
+  ret void
+}
+
+; i128 values weigh 4 registers each, so these three bodies peak at 144 with the
+; rewrites applied against 80 without. That straddles the budgets implied by the
+; occupancy attributes, letting the real TTI budget decide with no override.
+
+; No attributes: flat work group size defaults to 1024, so 4 waves per EU and a
+; 128-register budget. 144 exceeds the 115-register safe budget, so filtered.
+define void @default_occupancy(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) {
+; BUDGET32-LABEL: define void @default_occupancy(
+; BUDGET32-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) {
+; BUDGET32-NEXT:  [[ENTRY:.*:]]
+; BUDGET32-NEXT:    [[S2:%.*]] = shl i128 [[S]], 1
+; BUDGET32-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T1]])
+; BUDGET32-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T2]])
+; BUDGET32-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T3]])
+; BUDGET32-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T4]])
+; BUDGET32-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T5]])
+; BUDGET32-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T6]])
+; BUDGET32-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T7]])
+; BUDGET32-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T8]])
+; BUDGET32-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T9]])
+; BUDGET32-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T10]])
+; BUDGET32-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T11]])
+; BUDGET32-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T12]])
+; BUDGET32-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T13]])
+; BUDGET32-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T14]])
+; BUDGET32-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T15]])
+; BUDGET32-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T16]])
+; BUDGET32-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T17]])
+; BUDGET32-NEXT:    [[U1:%.*]] = add i128 [[B1]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U1]])
+; BUDGET32-NEXT:    [[U2:%.*]] = add i128 [[B2]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U2]])
+; BUDGET32-NEXT:    [[U3:%.*]] = add i128 [[B3]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U3]])
+; BUDGET32-NEXT:    [[U4:%.*]] = add i128 [[B4]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U4]])
+; BUDGET32-NEXT:    [[U5:%.*]] = add i128 [[B5]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U5]])
+; BUDGET32-NEXT:    [[U6:%.*]] = add i128 [[B6]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U6]])
+; BUDGET32-NEXT:    [[U7:%.*]] = add i128 [[B7]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U7]])
+; BUDGET32-NEXT:    [[U8:%.*]] = add i128 [[B8]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U8]])
+; BUDGET32-NEXT:    [[U9:%.*]] = add i128 [[B9]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U9]])
+; BUDGET32-NEXT:    [[U10:%.*]] = add i128 [[B10]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U10]])
+; BUDGET32-NEXT:    [[U11:%.*]] = add i128 [[B11]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U11]])
+; BUDGET32-NEXT:    [[U12:%.*]] = add i128 [[B12]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U12]])
+; BUDGET32-NEXT:    [[U13:%.*]] = add i128 [[B13]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U13]])
+; BUDGET32-NEXT:    [[U14:%.*]] = add i128 [[B14]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U14]])
+; BUDGET32-NEXT:    [[U15:%.*]] = add i128 [[B15]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U15]])
+; BUDGET32-NEXT:    [[U16:%.*]] = add i128 [[B16]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U16]])
+; BUDGET32-NEXT:    [[U17:%.*]] = add i128 [[B17]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U17]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B1]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B2]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B3]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B4]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B5]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B6]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B7]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B8]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B9]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B10]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B11]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B12]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B13]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B14]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B15]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B16]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B17]])
+; BUDGET32-NEXT:    ret void
+;
+; ATTRS-LABEL: define void @default_occupancy(
+; ATTRS-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) {
+; ATTRS-NEXT:  [[ENTRY:.*:]]
+; ATTRS-NEXT:    [[S2:%.*]] = shl i128 [[S]], 1
+; ATTRS-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T1]])
+; ATTRS-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T2]])
+; ATTRS-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T3]])
+; ATTRS-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T4]])
+; ATTRS-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T5]])
+; ATTRS-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T6]])
+; ATTRS-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T7]])
+; ATTRS-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T8]])
+; ATTRS-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T9]])
+; ATTRS-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T10]])
+; ATTRS-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T11]])
+; ATTRS-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T12]])
+; ATTRS-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T13]])
+; ATTRS-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T14]])
+; ATTRS-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T15]])
+; ATTRS-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T16]])
+; ATTRS-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T17]])
+; ATTRS-NEXT:    [[U1:%.*]] = add i128 [[B1]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U1]])
+; ATTRS-NEXT:    [[U2:%.*]] = add i128 [[B2]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U2]])
+; ATTRS-NEXT:    [[U3:%.*]] = add i128 [[B3]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U3]])
+; ATTRS-NEXT:    [[U4:%.*]] = add i128 [[B4]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U4]])
+; ATTRS-NEXT:    [[U5:%.*]] = add i128 [[B5]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U5]])
+; ATTRS-NEXT:    [[U6:%.*]] = add i128 [[B6]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U6]])
+; ATTRS-NEXT:    [[U7:%.*]] = add i128 [[B7]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U7]])
+; ATTRS-NEXT:    [[U8:%.*]] = add i128 [[B8]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U8]])
+; ATTRS-NEXT:    [[U9:%.*]] = add i128 [[B9]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U9]])
+; ATTRS-NEXT:    [[U10:%.*]] = add i128 [[B10]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U10]])
+; ATTRS-NEXT:    [[U11:%.*]] = add i128 [[B11]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U11]])
+; ATTRS-NEXT:    [[U12:%.*]] = add i128 [[B12]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U12]])
+; ATTRS-NEXT:    [[U13:%.*]] = add i128 [[B13]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U13]])
+; ATTRS-NEXT:    [[U14:%.*]] = add i128 [[B14]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U14]])
+; ATTRS-NEXT:    [[U15:%.*]] = add i128 [[B15]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U15]])
+; ATTRS-NEXT:    [[U16:%.*]] = add i128 [[B16]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U16]])
+; ATTRS-NEXT:    [[U17:%.*]] = add i128 [[B17]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U17]])
+; ATTRS-NEXT:    call void @bar(i128 [[B1]])
+; ATTRS-NEXT:    call void @bar(i128 [[B2]])
+; ATTRS-NEXT:    call void @bar(i128 [[B3]])
+; ATTRS-NEXT:    call void @bar(i128 [[B4]])
+; ATTRS-NEXT:    call void @bar(i128 [[B5]])
+; ATTRS-NEXT:    call void @bar(i128 [[B6]])
+; ATTRS-NEXT:    call void @bar(i128 [[B7]])
+; ATTRS-NEXT:    call void @bar(i128 [[B8]])
+; ATTRS-NEXT:    call void @bar(i128 [[B9]])
+; ATTRS-NEXT:    call void @bar(i128 [[B10]])
+; ATTRS-NEXT:    call void @bar(i128 [[B11]])
+; ATTRS-NEXT:    call void @bar(i128 [[B12]])
+; ATTRS-NEXT:    call void @bar(i128 [[B13]])
+; ATTRS-NEXT:    call void @bar(i128 [[B14]])
+; ATTRS-NEXT:    call void @bar(i128 [[B15]])
+; ATTRS-NEXT:    call void @bar(i128 [[B16]])
+; ATTRS-NEXT:    call void @bar(i128 [[B17]])
+; ATTRS-NEXT:    ret void
+;
+; NOFILTER-LABEL: define void @default_occupancy(
+; NOFILTER-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) {
+; NOFILTER-NEXT:  [[ENTRY:.*:]]
+; NOFILTER-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T1]])
+; NOFILTER-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T2]])
+; NOFILTER-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T3]])
+; NOFILTER-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T4]])
+; NOFILTER-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T5]])
+; NOFILTER-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T6]])
+; NOFILTER-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T7]])
+; NOFILTER-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T8]])
+; NOFILTER-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T9]])
+; NOFILTER-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T10]])
+; NOFILTER-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T11]])
+; NOFILTER-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T12]])
+; NOFILTER-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T13]])
+; NOFILTER-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T14]])
+; NOFILTER-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T15]])
+; NOFILTER-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T16]])
+; NOFILTER-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T17]])
+; NOFILTER-NEXT:    [[U1:%.*]] = add i128 [[T1]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U1]])
+; NOFILTER-NEXT:    [[U2:%.*]] = add i128 [[T2]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U2]])
+; NOFILTER-NEXT:    [[U3:%.*]] = add i128 [[T3]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U3]])
+; NOFILTER-NEXT:    [[U4:%.*]] = add i128 [[T4]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U4]])
+; NOFILTER-NEXT:    [[U5:%.*]] = add i128 [[T5]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U5]])
+; NOFILTER-NEXT:    [[U6:%.*]] = add i128 [[T6]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U6]])
+; NOFILTER-NEXT:    [[U7:%.*]] = add i128 [[T7]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U7]])
+; NOFILTER-NEXT:    [[U8:%.*]] = add i128 [[T8]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U8]])
+; NOFILTER-NEXT:    [[U9:%.*]] = add i128 [[T9]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U9]])
+; NOFILTER-NEXT:    [[U10:%.*]] = add i128 [[T10]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U10]])
+; NOFILTER-NEXT:    [[U11:%.*]] = add i128 [[T11]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U11]])
+; NOFILTER-NEXT:    [[U12:%.*]] = add i128 [[T12]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U12]])
+; NOFILTER-NEXT:    [[U13:%.*]] = add i128 [[T13]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U13]])
+; NOFILTER-NEXT:    [[U14:%.*]] = add i128 [[T14]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U14]])
+; NOFILTER-NEXT:    [[U15:%.*]] = add i128 [[T15]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U15]])
+; NOFILTER-NEXT:    [[U16:%.*]] = add i128 [[T16]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U16]])
+; NOFILTER-NEXT:    [[U17:%.*]] = add i128 [[T17]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U17]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B1]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B2]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B3]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B4]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B5]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B6]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B7]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B8]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B9]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B10]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B11]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B12]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B13]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B14]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B15]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B16]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B17]])
+; NOFILTER-NEXT:    ret void
+;
+entry:
+  %s2 = shl i128 %s, 1
+  %t1 = add i128 %b1, %s
+  call void @bar(i128 %t1)
+  %t2 = add i128 %b2, %s
+  call void @bar(i128 %t2)
+  %t3 = add i128 %b3, %s
+  call void @bar(i128 %t3)
+  %t4 = add i128 %b4, %s
+  call void @bar(i128 %t4)
+  %t5 = add i128 %b5, %s
+  call void @bar(i128 %t5)
+  %t6 = add i128 %b6, %s
+  call void @bar(i128 %t6)
+  %t7 = add i128 %b7, %s
+  call void @bar(i128 %t7)
+  %t8 = add i128 %b8, %s
+  call void @bar(i128 %t8)
+  %t9 = add i128 %b9, %s
+  call void @bar(i128 %t9)
+  %t10 = add i128 %b10, %s
+  call void @bar(i128 %t10)
+  %t11 = add i128 %b11, %s
+  call void @bar(i128 %t11)
+  %t12 = add i128 %b12, %s
+  call void @bar(i128 %t12)
+  %t13 = add i128 %b13, %s
+  call void @bar(i128 %t13)
+  %t14 = add i128 %b14, %s
+  call void @bar(i128 %t14)
+  %t15 = add i128 %b15, %s
+  call void @bar(i128 %t15)
+  %t16 = add i128 %b16, %s
+  call void @bar(i128 %t16)
+  %t17 = add i128 %b17, %s
+  call void @bar(i128 %t17)
+  %u1 = add i128 %b1, %s2
+  call void @bar(i128 %u1)
+  %u2 = add i128 %b2, %s2
+  call void @bar(i128 %u2)
+  %u3 = add i128 %b3, %s2
+  call void @bar(i128 %u3)
+  %u4 = add i128 %b4, %s2
+  call void @bar(i128 %u4)
+  %u5 = add i128 %b5, %s2
+  call void @bar(i128 %u5)
+  %u6 = add i128 %b6, %s2
+  call void @bar(i128 %u6)
+  %u7 = add i128 %b7, %s2
+  call void @bar(i128 %u7)
+  %u8 = add i128 %b8, %s2
+  call void @bar(i128 %u8)
+  %u9 = add i128 %b9, %s2
+  call void @bar(i128 %u9)
+  %u10 = add i128 %b10, %s2
+  call void @bar(i128 %u10)
+  %u11 = add i128 %b11, %s2
+  call void @bar(i128 %u11)
+  %u12 = add i128 %b12, %s2
+  call void @bar(i128 %u12)
+  %u13 = add i128 %b13, %s2
+  call void @bar(i128 %u13)
+  %u14 = add i128 %b14, %s2
+  call void @bar(i128 %u14)
+  %u15 = add i128 %b15, %s2
+  call void @bar(i128 %u15)
+  %u16 = add i128 %b16, %s2
+  call void @bar(i128 %u16)
+  %u17 = add i128 %b17, %s2
+  call void @bar(i128 %u17)
+  call void @bar(i128 %b1)
+  call void @bar(i128 %b2)
+  call void @bar(i128 %b3)
+  call void @bar(i128 %b4)
+  call void @bar(i128 %b5)
+  call void @bar(i128 %b6)
+  call void @bar(i128 %b7)
+  call void @bar(i128 %b8)
+  call void @bar(i128 %b9)
+  call void @bar(i128 %b10)
+  call void @bar(i128 %b11)
+  call void @bar(i128 %b12)
+  call void @bar(i128 %b13)
+  call void @bar(i128 %b14)
+  call void @bar(i128 %b15)
+  call void @bar(i128 %b16)
+  call void @bar(i128 %b17)
+  ret void
+}
+
+; A 256-thread work group is 4 wave64s over 4 EUs, so 1 wave per EU and the full
+; 512-register budget. The same 144 now fits, so the rewrites survive.
+define void @small_work_group(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) #0 {
+; BUDGET32-LABEL: define void @small_work_group(
+; BUDGET32-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR0:[0-9]+]] {
+; BUDGET32-NEXT:  [[ENTRY:.*:]]
+; BUDGET32-NEXT:    [[S2:%.*]] = shl i128 [[S]], 1
+; BUDGET32-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T1]])
+; BUDGET32-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T2]])
+; BUDGET32-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T3]])
+; BUDGET32-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T4]])
+; BUDGET32-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T5]])
+; BUDGET32-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T6]])
+; BUDGET32-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T7]])
+; BUDGET32-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T8]])
+; BUDGET32-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T9]])
+; BUDGET32-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T10]])
+; BUDGET32-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T11]])
+; BUDGET32-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T12]])
+; BUDGET32-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T13]])
+; BUDGET32-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T14]])
+; BUDGET32-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T15]])
+; BUDGET32-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T16]])
+; BUDGET32-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T17]])
+; BUDGET32-NEXT:    [[U1:%.*]] = add i128 [[B1]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U1]])
+; BUDGET32-NEXT:    [[U2:%.*]] = add i128 [[B2]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U2]])
+; BUDGET32-NEXT:    [[U3:%.*]] = add i128 [[B3]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U3]])
+; BUDGET32-NEXT:    [[U4:%.*]] = add i128 [[B4]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U4]])
+; BUDGET32-NEXT:    [[U5:%.*]] = add i128 [[B5]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U5]])
+; BUDGET32-NEXT:    [[U6:%.*]] = add i128 [[B6]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U6]])
+; BUDGET32-NEXT:    [[U7:%.*]] = add i128 [[B7]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U7]])
+; BUDGET32-NEXT:    [[U8:%.*]] = add i128 [[B8]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U8]])
+; BUDGET32-NEXT:    [[U9:%.*]] = add i128 [[B9]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U9]])
+; BUDGET32-NEXT:    [[U10:%.*]] = add i128 [[B10]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U10]])
+; BUDGET32-NEXT:    [[U11:%.*]] = add i128 [[B11]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U11]])
+; BUDGET32-NEXT:    [[U12:%.*]] = add i128 [[B12]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U12]])
+; BUDGET32-NEXT:    [[U13:%.*]] = add i128 [[B13]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U13]])
+; BUDGET32-NEXT:    [[U14:%.*]] = add i128 [[B14]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U14]])
+; BUDGET32-NEXT:    [[U15:%.*]] = add i128 [[B15]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U15]])
+; BUDGET32-NEXT:    [[U16:%.*]] = add i128 [[B16]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U16]])
+; BUDGET32-NEXT:    [[U17:%.*]] = add i128 [[B17]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U17]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B1]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B2]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B3]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B4]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B5]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B6]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B7]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B8]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B9]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B10]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B11]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B12]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B13]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B14]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B15]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B16]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B17]])
+; BUDGET32-NEXT:    ret void
+;
+; ATTRS-LABEL: define void @small_work_group(
+; ATTRS-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR0:[0-9]+]] {
+; ATTRS-NEXT:  [[ENTRY:.*:]]
+; ATTRS-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T1]])
+; ATTRS-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T2]])
+; ATTRS-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T3]])
+; ATTRS-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T4]])
+; ATTRS-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T5]])
+; ATTRS-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T6]])
+; ATTRS-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T7]])
+; ATTRS-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T8]])
+; ATTRS-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T9]])
+; ATTRS-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T10]])
+; ATTRS-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T11]])
+; ATTRS-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T12]])
+; ATTRS-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T13]])
+; ATTRS-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T14]])
+; ATTRS-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T15]])
+; ATTRS-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T16]])
+; ATTRS-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T17]])
+; ATTRS-NEXT:    [[U1:%.*]] = add i128 [[T1]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U1]])
+; ATTRS-NEXT:    [[U2:%.*]] = add i128 [[T2]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U2]])
+; ATTRS-NEXT:    [[U3:%.*]] = add i128 [[T3]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U3]])
+; ATTRS-NEXT:    [[U4:%.*]] = add i128 [[T4]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U4]])
+; ATTRS-NEXT:    [[U5:%.*]] = add i128 [[T5]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U5]])
+; ATTRS-NEXT:    [[U6:%.*]] = add i128 [[T6]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U6]])
+; ATTRS-NEXT:    [[U7:%.*]] = add i128 [[T7]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U7]])
+; ATTRS-NEXT:    [[U8:%.*]] = add i128 [[T8]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U8]])
+; ATTRS-NEXT:    [[U9:%.*]] = add i128 [[T9]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U9]])
+; ATTRS-NEXT:    [[U10:%.*]] = add i128 [[T10]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U10]])
+; ATTRS-NEXT:    [[U11:%.*]] = add i128 [[T11]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U11]])
+; ATTRS-NEXT:    [[U12:%.*]] = add i128 [[T12]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U12]])
+; ATTRS-NEXT:    [[U13:%.*]] = add i128 [[T13]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U13]])
+; ATTRS-NEXT:    [[U14:%.*]] = add i128 [[T14]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U14]])
+; ATTRS-NEXT:    [[U15:%.*]] = add i128 [[T15]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U15]])
+; ATTRS-NEXT:    [[U16:%.*]] = add i128 [[T16]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U16]])
+; ATTRS-NEXT:    [[U17:%.*]] = add i128 [[T17]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[U17]])
+; ATTRS-NEXT:    call void @bar(i128 [[B1]])
+; ATTRS-NEXT:    call void @bar(i128 [[B2]])
+; ATTRS-NEXT:    call void @bar(i128 [[B3]])
+; ATTRS-NEXT:    call void @bar(i128 [[B4]])
+; ATTRS-NEXT:    call void @bar(i128 [[B5]])
+; ATTRS-NEXT:    call void @bar(i128 [[B6]])
+; ATTRS-NEXT:    call void @bar(i128 [[B7]])
+; ATTRS-NEXT:    call void @bar(i128 [[B8]])
+; ATTRS-NEXT:    call void @bar(i128 [[B9]])
+; ATTRS-NEXT:    call void @bar(i128 [[B10]])
+; ATTRS-NEXT:    call void @bar(i128 [[B11]])
+; ATTRS-NEXT:    call void @bar(i128 [[B12]])
+; ATTRS-NEXT:    call void @bar(i128 [[B13]])
+; ATTRS-NEXT:    call void @bar(i128 [[B14]])
+; ATTRS-NEXT:    call void @bar(i128 [[B15]])
+; ATTRS-NEXT:    call void @bar(i128 [[B16]])
+; ATTRS-NEXT:    call void @bar(i128 [[B17]])
+; ATTRS-NEXT:    ret void
+;
+; NOFILTER-LABEL: define void @small_work_group(
+; NOFILTER-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR0:[0-9]+]] {
+; NOFILTER-NEXT:  [[ENTRY:.*:]]
+; NOFILTER-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T1]])
+; NOFILTER-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T2]])
+; NOFILTER-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T3]])
+; NOFILTER-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T4]])
+; NOFILTER-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T5]])
+; NOFILTER-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T6]])
+; NOFILTER-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T7]])
+; NOFILTER-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T8]])
+; NOFILTER-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T9]])
+; NOFILTER-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T10]])
+; NOFILTER-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T11]])
+; NOFILTER-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T12]])
+; NOFILTER-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T13]])
+; NOFILTER-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T14]])
+; NOFILTER-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T15]])
+; NOFILTER-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T16]])
+; NOFILTER-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T17]])
+; NOFILTER-NEXT:    [[U1:%.*]] = add i128 [[T1]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U1]])
+; NOFILTER-NEXT:    [[U2:%.*]] = add i128 [[T2]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U2]])
+; NOFILTER-NEXT:    [[U3:%.*]] = add i128 [[T3]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U3]])
+; NOFILTER-NEXT:    [[U4:%.*]] = add i128 [[T4]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U4]])
+; NOFILTER-NEXT:    [[U5:%.*]] = add i128 [[T5]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U5]])
+; NOFILTER-NEXT:    [[U6:%.*]] = add i128 [[T6]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U6]])
+; NOFILTER-NEXT:    [[U7:%.*]] = add i128 [[T7]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U7]])
+; NOFILTER-NEXT:    [[U8:%.*]] = add i128 [[T8]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U8]])
+; NOFILTER-NEXT:    [[U9:%.*]] = add i128 [[T9]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U9]])
+; NOFILTER-NEXT:    [[U10:%.*]] = add i128 [[T10]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U10]])
+; NOFILTER-NEXT:    [[U11:%.*]] = add i128 [[T11]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U11]])
+; NOFILTER-NEXT:    [[U12:%.*]] = add i128 [[T12]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U12]])
+; NOFILTER-NEXT:    [[U13:%.*]] = add i128 [[T13]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U13]])
+; NOFILTER-NEXT:    [[U14:%.*]] = add i128 [[T14]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U14]])
+; NOFILTER-NEXT:    [[U15:%.*]] = add i128 [[T15]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U15]])
+; NOFILTER-NEXT:    [[U16:%.*]] = add i128 [[T16]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U16]])
+; NOFILTER-NEXT:    [[U17:%.*]] = add i128 [[T17]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U17]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B1]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B2]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B3]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B4]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B5]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B6]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B7]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B8]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B9]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B10]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B11]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B12]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B13]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B14]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B15]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B16]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B17]])
+; NOFILTER-NEXT:    ret void
+;
+entry:
+  %s2 = shl i128 %s, 1
+  %t1 = add i128 %b1, %s
+  call void @bar(i128 %t1)
+  %t2 = add i128 %b2, %s
+  call void @bar(i128 %t2)
+  %t3 = add i128 %b3, %s
+  call void @bar(i128 %t3)
+  %t4 = add i128 %b4, %s
+  call void @bar(i128 %t4)
+  %t5 = add i128 %b5, %s
+  call void @bar(i128 %t5)
+  %t6 = add i128 %b6, %s
+  call void @bar(i128 %t6)
+  %t7 = add i128 %b7, %s
+  call void @bar(i128 %t7)
+  %t8 = add i128 %b8, %s
+  call void @bar(i128 %t8)
+  %t9 = add i128 %b9, %s
+  call void @bar(i128 %t9)
+  %t10 = add i128 %b10, %s
+  call void @bar(i128 %t10)
+  %t11 = add i128 %b11, %s
+  call void @bar(i128 %t11)
+  %t12 = add i128 %b12, %s
+  call void @bar(i128 %t12)
+  %t13 = add i128 %b13, %s
+  call void @bar(i128 %t13)
+  %t14 = add i128 %b14, %s
+  call void @bar(i128 %t14)
+  %t15 = add i128 %b15, %s
+  call void @bar(i128 %t15)
+  %t16 = add i128 %b16, %s
+  call void @bar(i128 %t16)
+  %t17 = add i128 %b17, %s
+  call void @bar(i128 %t17)
+  %u1 = add i128 %b1, %s2
+  call void @bar(i128 %u1)
+  %u2 = add i128 %b2, %s2
+  call void @bar(i128 %u2)
+  %u3 = add i128 %b3, %s2
+  call void @bar(i128 %u3)
+  %u4 = add i128 %b4, %s2
+  call void @bar(i128 %u4)
+  %u5 = add i128 %b5, %s2
+  call void @bar(i128 %u5)
+  %u6 = add i128 %b6, %s2
+  call void @bar(i128 %u6)
+  %u7 = add i128 %b7, %s2
+  call void @bar(i128 %u7)
+  %u8 = add i128 %b8, %s2
+  call void @bar(i128 %u8)
+  %u9 = add i128 %b9, %s2
+  call void @bar(i128 %u9)
+  %u10 = add i128 %b10, %s2
+  call void @bar(i128 %u10)
+  %u11 = add i128 %b11, %s2
+  call void @bar(i128 %u11)
+  %u12 = add i128 %b12, %s2
+  call void @bar(i128 %u12)
+  %u13 = add i128 %b13, %s2
+  call void @bar(i128 %u13)
+  %u14 = add i128 %b14, %s2
+  call void @bar(i128 %u14)
+  %u15 = add i128 %b15, %s2
+  call void @bar(i128 %u15)
+  %u16 = add i128 %b16, %s2
+  call void @bar(i128 %u16)
+  %u17 = add i128 %b17, %s2
+  call void @bar(i128 %u17)
+  call void @bar(i128 %b1)
+  call void @bar(i128 %b2)
+  call void @bar(i128 %b3)
+  call void @bar(i128 %b4)
+  call void @bar(i128 %b5)
+  call void @bar(i128 %b6)
+  call void @bar(i128 %b7)
+  call void @bar(i128 %b8)
+  call void @bar(i128 %b9)
+  call void @bar(i128 %b10)
+  call void @bar(i128 %b11)
+  call void @bar(i128 %b12)
+  call void @bar(i128 %b13)
+  call void @bar(i128 %b14)
+  call void @bar(i128 %b15)
+  call void @bar(i128 %b16)
+  call void @bar(i128 %b17)
+  ret void
+}
+
+; Same small work group, but asking for 8 waves per EU drops the budget to
+; 512/8 = 64, so 144 is over again and the rewrites are dropped. This is the
+; pair that shows waves-per-eu overriding what the work group size allows.
+define void @small_wg_high_occ(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) #1 {
+; BUDGET32-LABEL: define void @small_wg_high_occ(
+; BUDGET32-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR1:[0-9]+]] {
+; BUDGET32-NEXT:  [[ENTRY:.*:]]
+; BUDGET32-NEXT:    [[S2:%.*]] = shl i128 [[S]], 1
+; BUDGET32-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T1]])
+; BUDGET32-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T2]])
+; BUDGET32-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T3]])
+; BUDGET32-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T4]])
+; BUDGET32-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T5]])
+; BUDGET32-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T6]])
+; BUDGET32-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T7]])
+; BUDGET32-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T8]])
+; BUDGET32-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T9]])
+; BUDGET32-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T10]])
+; BUDGET32-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T11]])
+; BUDGET32-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T12]])
+; BUDGET32-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T13]])
+; BUDGET32-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T14]])
+; BUDGET32-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T15]])
+; BUDGET32-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T16]])
+; BUDGET32-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
+; BUDGET32-NEXT:    call void @bar(i128 [[T17]])
+; BUDGET32-NEXT:    [[U1:%.*]] = add i128 [[B1]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U1]])
+; BUDGET32-NEXT:    [[U2:%.*]] = add i128 [[B2]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U2]])
+; BUDGET32-NEXT:    [[U3:%.*]] = add i128 [[B3]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U3]])
+; BUDGET32-NEXT:    [[U4:%.*]] = add i128 [[B4]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U4]])
+; BUDGET32-NEXT:    [[U5:%.*]] = add i128 [[B5]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U5]])
+; BUDGET32-NEXT:    [[U6:%.*]] = add i128 [[B6]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U6]])
+; BUDGET32-NEXT:    [[U7:%.*]] = add i128 [[B7]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U7]])
+; BUDGET32-NEXT:    [[U8:%.*]] = add i128 [[B8]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U8]])
+; BUDGET32-NEXT:    [[U9:%.*]] = add i128 [[B9]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U9]])
+; BUDGET32-NEXT:    [[U10:%.*]] = add i128 [[B10]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U10]])
+; BUDGET32-NEXT:    [[U11:%.*]] = add i128 [[B11]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U11]])
+; BUDGET32-NEXT:    [[U12:%.*]] = add i128 [[B12]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U12]])
+; BUDGET32-NEXT:    [[U13:%.*]] = add i128 [[B13]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U13]])
+; BUDGET32-NEXT:    [[U14:%.*]] = add i128 [[B14]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U14]])
+; BUDGET32-NEXT:    [[U15:%.*]] = add i128 [[B15]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U15]])
+; BUDGET32-NEXT:    [[U16:%.*]] = add i128 [[B16]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U16]])
+; BUDGET32-NEXT:    [[U17:%.*]] = add i128 [[B17]], [[S2]]
+; BUDGET32-NEXT:    call void @bar(i128 [[U17]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B1]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B2]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B3]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B4]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B5]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B6]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B7]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B8]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B9]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B10]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B11]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B12]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B13]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B14]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B15]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B16]])
+; BUDGET32-NEXT:    call void @bar(i128 [[B17]])
+; BUDGET32-NEXT:    ret void
+;
+; ATTRS-LABEL: define void @small_wg_high_occ(
+; ATTRS-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR1:[0-9]+]] {
+; ATTRS-NEXT:  [[ENTRY:.*:]]
+; ATTRS-NEXT:    [[S2:%.*]] = shl i128 [[S]], 1
+; ATTRS-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T1]])
+; ATTRS-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T2]])
+; ATTRS-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T3]])
+; ATTRS-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T4]])
+; ATTRS-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T5]])
+; ATTRS-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T6]])
+; ATTRS-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T7]])
+; ATTRS-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T8]])
+; ATTRS-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T9]])
+; ATTRS-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T10]])
+; ATTRS-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T11]])
+; ATTRS-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T12]])
+; ATTRS-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T13]])
+; ATTRS-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T14]])
+; ATTRS-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T15]])
+; ATTRS-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T16]])
+; ATTRS-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
+; ATTRS-NEXT:    call void @bar(i128 [[T17]])
+; ATTRS-NEXT:    [[U1:%.*]] = add i128 [[B1]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U1]])
+; ATTRS-NEXT:    [[U2:%.*]] = add i128 [[B2]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U2]])
+; ATTRS-NEXT:    [[U3:%.*]] = add i128 [[B3]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U3]])
+; ATTRS-NEXT:    [[U4:%.*]] = add i128 [[B4]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U4]])
+; ATTRS-NEXT:    [[U5:%.*]] = add i128 [[B5]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U5]])
+; ATTRS-NEXT:    [[U6:%.*]] = add i128 [[B6]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U6]])
+; ATTRS-NEXT:    [[U7:%.*]] = add i128 [[B7]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U7]])
+; ATTRS-NEXT:    [[U8:%.*]] = add i128 [[B8]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U8]])
+; ATTRS-NEXT:    [[U9:%.*]] = add i128 [[B9]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U9]])
+; ATTRS-NEXT:    [[U10:%.*]] = add i128 [[B10]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U10]])
+; ATTRS-NEXT:    [[U11:%.*]] = add i128 [[B11]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U11]])
+; ATTRS-NEXT:    [[U12:%.*]] = add i128 [[B12]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U12]])
+; ATTRS-NEXT:    [[U13:%.*]] = add i128 [[B13]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U13]])
+; ATTRS-NEXT:    [[U14:%.*]] = add i128 [[B14]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U14]])
+; ATTRS-NEXT:    [[U15:%.*]] = add i128 [[B15]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U15]])
+; ATTRS-NEXT:    [[U16:%.*]] = add i128 [[B16]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U16]])
+; ATTRS-NEXT:    [[U17:%.*]] = add i128 [[B17]], [[S2]]
+; ATTRS-NEXT:    call void @bar(i128 [[U17]])
+; ATTRS-NEXT:    call void @bar(i128 [[B1]])
+; ATTRS-NEXT:    call void @bar(i128 [[B2]])
+; ATTRS-NEXT:    call void @bar(i128 [[B3]])
+; ATTRS-NEXT:    call void @bar(i128 [[B4]])
+; ATTRS-NEXT:    call void @bar(i128 [[B5]])
+; ATTRS-NEXT:    call void @bar(i128 [[B6]])
+; ATTRS-NEXT:    call void @bar(i128 [[B7]])
+; ATTRS-NEXT:    call void @bar(i128 [[B8]])
+; ATTRS-NEXT:    call void @bar(i128 [[B9]])
+; ATTRS-NEXT:    call void @bar(i128 [[B10]])
+; ATTRS-NEXT:    call void @bar(i128 [[B11]])
+; ATTRS-NEXT:    call void @bar(i128 [[B12]])
+; ATTRS-NEXT:    call void @bar(i128 [[B13]])
+; ATTRS-NEXT:    call void @bar(i128 [[B14]])
+; ATTRS-NEXT:    call void @bar(i128 [[B15]])
+; ATTRS-NEXT:    call void @bar(i128 [[B16]])
+; ATTRS-NEXT:    call void @bar(i128 [[B17]])
+; ATTRS-NEXT:    ret void
+;
+; NOFILTER-LABEL: define void @small_wg_high_occ(
+; NOFILTER-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR1:[0-9]+]] {
+; NOFILTER-NEXT:  [[ENTRY:.*:]]
+; NOFILTER-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T1]])
+; NOFILTER-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T2]])
+; NOFILTER-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T3]])
+; NOFILTER-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T4]])
+; NOFILTER-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T5]])
+; NOFILTER-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T6]])
+; NOFILTER-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T7]])
+; NOFILTER-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T8]])
+; NOFILTER-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T9]])
+; NOFILTER-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T10]])
+; NOFILTER-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T11]])
+; NOFILTER-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T12]])
+; NOFILTER-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T13]])
+; NOFILTER-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T14]])
+; NOFILTER-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T15]])
+; NOFILTER-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T16]])
+; NOFILTER-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[T17]])
+; NOFILTER-NEXT:    [[U1:%.*]] = add i128 [[T1]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U1]])
+; NOFILTER-NEXT:    [[U2:%.*]] = add i128 [[T2]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U2]])
+; NOFILTER-NEXT:    [[U3:%.*]] = add i128 [[T3]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U3]])
+; NOFILTER-NEXT:    [[U4:%.*]] = add i128 [[T4]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U4]])
+; NOFILTER-NEXT:    [[U5:%.*]] = add i128 [[T5]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U5]])
+; NOFILTER-NEXT:    [[U6:%.*]] = add i128 [[T6]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U6]])
+; NOFILTER-NEXT:    [[U7:%.*]] = add i128 [[T7]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U7]])
+; NOFILTER-NEXT:    [[U8:%.*]] = add i128 [[T8]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U8]])
+; NOFILTER-NEXT:    [[U9:%.*]] = add i128 [[T9]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U9]])
+; NOFILTER-NEXT:    [[U10:%.*]] = add i128 [[T10]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U10]])
+; NOFILTER-NEXT:    [[U11:%.*]] = add i128 [[T11]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U11]])
+; NOFILTER-NEXT:    [[U12:%.*]] = add i128 [[T12]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U12]])
+; NOFILTER-NEXT:    [[U13:%.*]] = add i128 [[T13]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U13]])
+; NOFILTER-NEXT:    [[U14:%.*]] = add i128 [[T14]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U14]])
+; NOFILTER-NEXT:    [[U15:%.*]] = add i128 [[T15]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U15]])
+; NOFILTER-NEXT:    [[U16:%.*]] = add i128 [[T16]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U16]])
+; NOFILTER-NEXT:    [[U17:%.*]] = add i128 [[T17]], [[S]]
+; NOFILTER-NEXT:    call void @bar(i128 [[U17]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B1]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B2]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B3]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B4]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B5]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B6]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B7]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B8]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B9]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B10]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B11]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B12]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B13]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B14]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B15]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B16]])
+; NOFILTER-NEXT:    call void @bar(i128 [[B17]])
+; NOFILTER-NEXT:    ret void
+;
+entry:
+  %s2 = shl i128 %s, 1
+  %t1 = add i128 %b1, %s
+  call void @bar(i128 %t1)
+  %t2 = add i128 %b2, %s
+  call void @bar(i128 %t2)
+  %t3 = add i128 %b3, %s
+  call void @bar(i128 %t3)
+  %t4 = add i128 %b4, %s
+  call void @bar(i128 %t4)
+  %t5 = add i128 %b5, %s
+  call void @bar(i128 %t5)
+  %t6 = add i128 %b6, %s
+  call void @bar(i128 %t6)
+  %t7 = add i128 %b7, %s
+  call void @bar(i128 %t7)
+  %t8 = add i128 %b8, %s
+  call void @bar(i128 %t8)
+  %t9 = add i128 %b9, %s
+  call void @bar(i128 %t9)
+  %t10 = add i128 %b10, %s
+  call void @bar(i128 %t10)
+  %t11 = add i128 %b11, %s
+  call void @bar(i128 %t11)
+  %t12 = add i128 %b12, %s
+  call void @bar(i128 %t12)
+  %t13 = add i128 %b13, %s
+  call void @bar(i128 %t13)
+  %t14 = add i128 %b14, %s
+  call void @bar(i128 %t14)
+  %t15 = add i128 %b15, %s
+  call void @bar(i128 %t15)
+  %t16 = add i128 %b16, %s
+  call void @bar(i128 %t16)
+  %t17 = add i128 %b17, %s
+  call void @bar(i128 %t17)
+  %u1 = add i128 %b1, %s2
+  call void @bar(i128 %u1)
+  %u2 = add i128 %b2, %s2
+  call void @bar(i128 %u2)
+  %u3 = add i128 %b3, %s2
+  call void @bar(i128 %u3)
+  %u4 = add i128 %b4, %s2
+  call void @bar(i128 %u4)
+  %u5 = add i128 %b5, %s2
+  call void @bar(i128 %u5)
+  %u6 = add i128 %b6, %s2
+  call void @bar(i128 %u6)
+  %u7 = add i128 %b7, %s2
+  call void @bar(i128 %u7)
+  %u8 = add i128 %b8, %s2
+  call void @bar(i128 %u8)
+  %u9 = add i128 %b9, %s2
+  call void @bar(i128 %u9)
+  %u10 = add i128 %b10, %s2
+  call void @bar(i128 %u10)
+  %u11 = add i128 %b11, %s2
+  call void @bar(i128 %u11)
+  %u12 = add i128 %b12, %s2
+  call void @bar(i128 %u12)
+  %u13 = add i128 %b13, %s2
+  call void @bar(i128 %u13)
+  %u14 = add i128 %b14, %s2
+  call void @bar(i128 %u14)
+  %u15 = add i128 %b15, %s2
+  call void @bar(i128 %u15)
+  %u16 = add i128 %b16, %s2
+  call void @bar(i128 %u16)
+  %u17 = add i128 %b17, %s2
+  call void @bar(i128 %u17)
+  call void @bar(i128 %b1)
+  call void @bar(i128 %b2)
+  call void @bar(i128 %b3)
+  call void @bar(i128 %b4)
+  call void @bar(i128 %b5)
+  call void @bar(i128 %b6)
+  call void @bar(i128 %b7)
+  call void @bar(i128 %b8)
+  call void @bar(i128 %b9)
+  call void @bar(i128 %b10)
+  call void @bar(i128 %b11)
+  call void @bar(i128 %b12)
+  call void @bar(i128 %b13)
+  call void @bar(i128 %b14)
+  call void @bar(i128 %b15)
+  call void @bar(i128 %b16)
+  call void @bar(i128 %b17)
+  ret void
+}
+
+attributes #0 = { "amdgpu-flat-work-group-size"="1,256" }
+attributes #1 = { "amdgpu-flat-work-group-size"="1,256" "amdgpu-waves-per-eu"="8" }
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
deleted file mode 100644
index 42aa95b8dd006..0000000000000
--- a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-basis-distance-threshold.ll
+++ /dev/null
@@ -1,94 +0,0 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt < %s -passes=slsr -S -slsr-basis-distance-threshold=96 | FileCheck %s --check-prefixes=CHECK,REWRITE
-; RUN: opt < %s -passes=slsr -S -slsr-basis-distance-threshold=2  | FileCheck %s --check-prefixes=CHECK,SKIP
-
-; The register-pressure cost model skips a rewrite only when BOTH:
-;   1. an operand of the candidate is used by a non-rewritable user in
-;      another block, and
-;   2. the basis' last same-block use is farther than
-;      -slsr-basis-distance-threshold from the candidate.
-;
-; Rewriting of %t2 = add i32 %t1, %s based on basis %t1 shouldn't happen when slsr-basis-distance-threshold is 2.
-; Distance from %t1's last use to %t2 is 5 > 2 and %s2 continues to be used in "next".
-
-target datalayout = "e-i64:64-v16:16-v32:32-n16:32:64"
-
-declare void @foo(i32)
-declare void @use(i32)
-
-define void @basis_too_far(i32 %b, i32 %s) {
-; REWRITE-LABEL: define void @basis_too_far(
-; REWRITE-SAME: i32 [[B:%.*]], i32 [[S:%.*]]) {
-; REWRITE-NEXT:  [[ENTRY:.*:]]
-; REWRITE-NEXT:    [[T1:%.*]] = add i32 [[B]], [[S]]
-; REWRITE-NEXT:    call void @foo(i32 [[T1]])
-; REWRITE-NEXT:    call void @foo(i32 [[B]])
-; REWRITE-NEXT:    call void @foo(i32 [[B]])
-; REWRITE-NEXT:    call void @foo(i32 [[B]])
-; REWRITE-NEXT:    [[S2:%.*]] = shl i32 [[S]], 1
-; REWRITE-NEXT:    [[T2:%.*]] = add i32 [[T1]], [[S]]
-; REWRITE-NEXT:    call void @foo(i32 [[T2]])
-; REWRITE-NEXT:    br label %[[NEXT:.*]]
-; REWRITE:       [[NEXT]]:
-; REWRITE-NEXT:    call void @use(i32 [[S2]])
-; REWRITE-NEXT:    ret void
-;
-; SKIP-LABEL: define void @basis_too_far(
-; SKIP-SAME: i32 [[B:%.*]], i32 [[S:%.*]]) {
-; SKIP-NEXT:  [[ENTRY:.*:]]
-; SKIP-NEXT:    [[T1:%.*]] = add i32 [[B]], [[S]]
-; SKIP-NEXT:    call void @foo(i32 [[T1]])
-; SKIP-NEXT:    call void @foo(i32 [[B]])
-; SKIP-NEXT:    call void @foo(i32 [[B]])
-; SKIP-NEXT:    call void @foo(i32 [[B]])
-; SKIP-NEXT:    [[S2:%.*]] = shl i32 [[S]], 1
-; SKIP-NEXT:    [[T2:%.*]] = add i32 [[B]], [[S2]]
-; SKIP-NEXT:    call void @foo(i32 [[T2]])
-; SKIP-NEXT:    br label %[[NEXT:.*]]
-; SKIP:       [[NEXT]]:
-; SKIP-NEXT:    call void @use(i32 [[S2]])
-; SKIP-NEXT:    ret void
-;
-entry:
-  %t1 = add i32 %b, %s
-  call void @foo(i32 %t1)
-  call void @foo(i32 %b)
-  call void @foo(i32 %b)
-  call void @foo(i32 %b)
-  %s2 = shl i32 %s, 1
-  %t2 = add i32 %b, %s2
-  call void @foo(i32 %t2)
-  br label %next
-
-next:
-  call void @use(i32 %s2)
-  ret void
-}
-
-define void @same_block_operand(i32 %b, i32 %s) {
-; CHECK-LABEL: define void @same_block_operand(
-; CHECK-SAME: i32 [[B:%.*]], i32 [[S:%.*]]) {
-; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[T1:%.*]] = add i32 [[B]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T1]])
-; CHECK-NEXT:    call void @foo(i32 [[B]])
-; CHECK-NEXT:    call void @foo(i32 [[B]])
-; CHECK-NEXT:    call void @foo(i32 [[B]])
-; CHECK-NEXT:    [[S2:%.*]] = shl i32 [[S]], 1
-; CHECK-NEXT:    [[T2:%.*]] = add i32 [[T1]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T2]])
-; CHECK-NEXT:    call void @use(i32 [[S2]])
-; CHECK-NEXT:    ret void
-;
-entry:
-  %t1 = add i32 %b, %s
-  call void @foo(i32 %t1)
-  call void @foo(i32 %b)
-  call void @foo(i32 %b)
-  call void @foo(i32 %b)
-  %s2 = shl i32 %s, 1
-  %t2 = add i32 %b, %s2
-  call void @foo(i32 %t2)
-  call void @use(i32 %s2)
-  ret void
-}
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
new file mode 100644
index 0000000000000..c534ecb8a3fda
--- /dev/null
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
@@ -0,0 +1,296 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=slsr -S | FileCheck %s --check-prefixes=CHECK,NOBUDGET
+; RUN: opt < %s -passes=slsr -S -slsr-reg-budget=32 | FileCheck %s --check-prefixes=CHECK,BUDGET
+
+; The register-pressure filter needs a register budget to compare against, and
+; the generic TargetTransformInfo has none: getRegisterBudget() returns
+; std::nullopt unless a target implements it. There is no target triple here, so
+; the filter returns early without even running its liveness or pressure
+; analyses, and every rewrite stands no matter how much pressure it adds.
+;
+; @many_bases_overlapping is the same shape used in AMDGPU/slsr-rp-filter.ll:
+; 17 distinct bases whose rewrites take the block's peak pressure from 20 to 36
+; registers. Passing -slsr-reg-budget=32 supplies the missing budget, and with
+; the 0.9 safe fraction leaving 28 registers the rewrites are then dropped.
+;
+; So the two runs differ only in whether a budget exists, which is what pins the
+; early return down: without one the filter is inert on targets that cannot
+; report a register count.
+
+target datalayout = "e-i64:64-v16:16-v32:32-n16:32:64"
+
+declare void @foo(i32)
+
+define void @many_bases_overlapping(i32 %s, i32 %b1, i32 %b2, i32 %b3, i32 %b4, i32 %b5, i32 %b6, i32 %b7, i32 %b8, i32 %b9, i32 %b10, i32 %b11, i32 %b12, i32 %b13, i32 %b14, i32 %b15, i32 %b16, i32 %b17) {
+; NOBUDGET-LABEL: define void @many_bases_overlapping(
+; NOBUDGET-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; NOBUDGET-NEXT:  [[ENTRY:.*:]]
+; NOBUDGET-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T1]])
+; NOBUDGET-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T2]])
+; NOBUDGET-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T3]])
+; NOBUDGET-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T4]])
+; NOBUDGET-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T5]])
+; NOBUDGET-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T6]])
+; NOBUDGET-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T7]])
+; NOBUDGET-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T8]])
+; NOBUDGET-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T9]])
+; NOBUDGET-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T10]])
+; NOBUDGET-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T11]])
+; NOBUDGET-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T12]])
+; NOBUDGET-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T13]])
+; NOBUDGET-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T14]])
+; NOBUDGET-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T15]])
+; NOBUDGET-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T16]])
+; NOBUDGET-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[T17]])
+; NOBUDGET-NEXT:    [[U1:%.*]] = add i32 [[T1]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U1]])
+; NOBUDGET-NEXT:    [[U2:%.*]] = add i32 [[T2]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U2]])
+; NOBUDGET-NEXT:    [[U3:%.*]] = add i32 [[T3]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U3]])
+; NOBUDGET-NEXT:    [[U4:%.*]] = add i32 [[T4]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U4]])
+; NOBUDGET-NEXT:    [[U5:%.*]] = add i32 [[T5]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U5]])
+; NOBUDGET-NEXT:    [[U6:%.*]] = add i32 [[T6]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U6]])
+; NOBUDGET-NEXT:    [[U7:%.*]] = add i32 [[T7]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U7]])
+; NOBUDGET-NEXT:    [[U8:%.*]] = add i32 [[T8]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U8]])
+; NOBUDGET-NEXT:    [[U9:%.*]] = add i32 [[T9]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U9]])
+; NOBUDGET-NEXT:    [[U10:%.*]] = add i32 [[T10]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U10]])
+; NOBUDGET-NEXT:    [[U11:%.*]] = add i32 [[T11]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U11]])
+; NOBUDGET-NEXT:    [[U12:%.*]] = add i32 [[T12]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U12]])
+; NOBUDGET-NEXT:    [[U13:%.*]] = add i32 [[T13]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U13]])
+; NOBUDGET-NEXT:    [[U14:%.*]] = add i32 [[T14]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U14]])
+; NOBUDGET-NEXT:    [[U15:%.*]] = add i32 [[T15]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U15]])
+; NOBUDGET-NEXT:    [[U16:%.*]] = add i32 [[T16]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U16]])
+; NOBUDGET-NEXT:    [[U17:%.*]] = add i32 [[T17]], [[S]]
+; NOBUDGET-NEXT:    call void @foo(i32 [[U17]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B1]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B2]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B3]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B4]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B5]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B6]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B7]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B8]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B9]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B10]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B11]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B12]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B13]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B14]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B15]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B16]])
+; NOBUDGET-NEXT:    call void @foo(i32 [[B17]])
+; NOBUDGET-NEXT:    ret void
+;
+; BUDGET-LABEL: define void @many_bases_overlapping(
+; BUDGET-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
+; BUDGET-NEXT:  [[ENTRY:.*:]]
+; BUDGET-NEXT:    [[S2:%.*]] = shl i32 [[S]], 1
+; BUDGET-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T1]])
+; BUDGET-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T2]])
+; BUDGET-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T3]])
+; BUDGET-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T4]])
+; BUDGET-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T5]])
+; BUDGET-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T6]])
+; BUDGET-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T7]])
+; BUDGET-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T8]])
+; BUDGET-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T9]])
+; BUDGET-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T10]])
+; BUDGET-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T11]])
+; BUDGET-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T12]])
+; BUDGET-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T13]])
+; BUDGET-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T14]])
+; BUDGET-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T15]])
+; BUDGET-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T16]])
+; BUDGET-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
+; BUDGET-NEXT:    call void @foo(i32 [[T17]])
+; BUDGET-NEXT:    [[U1:%.*]] = add i32 [[B1]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U1]])
+; BUDGET-NEXT:    [[U2:%.*]] = add i32 [[B2]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U2]])
+; BUDGET-NEXT:    [[U3:%.*]] = add i32 [[B3]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U3]])
+; BUDGET-NEXT:    [[U4:%.*]] = add i32 [[B4]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U4]])
+; BUDGET-NEXT:    [[U5:%.*]] = add i32 [[B5]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U5]])
+; BUDGET-NEXT:    [[U6:%.*]] = add i32 [[B6]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U6]])
+; BUDGET-NEXT:    [[U7:%.*]] = add i32 [[B7]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U7]])
+; BUDGET-NEXT:    [[U8:%.*]] = add i32 [[B8]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U8]])
+; BUDGET-NEXT:    [[U9:%.*]] = add i32 [[B9]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U9]])
+; BUDGET-NEXT:    [[U10:%.*]] = add i32 [[B10]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U10]])
+; BUDGET-NEXT:    [[U11:%.*]] = add i32 [[B11]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U11]])
+; BUDGET-NEXT:    [[U12:%.*]] = add i32 [[B12]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U12]])
+; BUDGET-NEXT:    [[U13:%.*]] = add i32 [[B13]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U13]])
+; BUDGET-NEXT:    [[U14:%.*]] = add i32 [[B14]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U14]])
+; BUDGET-NEXT:    [[U15:%.*]] = add i32 [[B15]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U15]])
+; BUDGET-NEXT:    [[U16:%.*]] = add i32 [[B16]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U16]])
+; BUDGET-NEXT:    [[U17:%.*]] = add i32 [[B17]], [[S2]]
+; BUDGET-NEXT:    call void @foo(i32 [[U17]])
+; BUDGET-NEXT:    call void @foo(i32 [[B1]])
+; BUDGET-NEXT:    call void @foo(i32 [[B2]])
+; BUDGET-NEXT:    call void @foo(i32 [[B3]])
+; BUDGET-NEXT:    call void @foo(i32 [[B4]])
+; BUDGET-NEXT:    call void @foo(i32 [[B5]])
+; BUDGET-NEXT:    call void @foo(i32 [[B6]])
+; BUDGET-NEXT:    call void @foo(i32 [[B7]])
+; BUDGET-NEXT:    call void @foo(i32 [[B8]])
+; BUDGET-NEXT:    call void @foo(i32 [[B9]])
+; BUDGET-NEXT:    call void @foo(i32 [[B10]])
+; BUDGET-NEXT:    call void @foo(i32 [[B11]])
+; BUDGET-NEXT:    call void @foo(i32 [[B12]])
+; BUDGET-NEXT:    call void @foo(i32 [[B13]])
+; BUDGET-NEXT:    call void @foo(i32 [[B14]])
+; BUDGET-NEXT:    call void @foo(i32 [[B15]])
+; BUDGET-NEXT:    call void @foo(i32 [[B16]])
+; BUDGET-NEXT:    call void @foo(i32 [[B17]])
+; BUDGET-NEXT:    ret void
+;
+entry:
+  %s2 = shl i32 %s, 1
+  %t1 = add i32 %b1, %s
+  call void @foo(i32 %t1)
+  %t2 = add i32 %b2, %s
+  call void @foo(i32 %t2)
+  %t3 = add i32 %b3, %s
+  call void @foo(i32 %t3)
+  %t4 = add i32 %b4, %s
+  call void @foo(i32 %t4)
+  %t5 = add i32 %b5, %s
+  call void @foo(i32 %t5)
+  %t6 = add i32 %b6, %s
+  call void @foo(i32 %t6)
+  %t7 = add i32 %b7, %s
+  call void @foo(i32 %t7)
+  %t8 = add i32 %b8, %s
+  call void @foo(i32 %t8)
+  %t9 = add i32 %b9, %s
+  call void @foo(i32 %t9)
+  %t10 = add i32 %b10, %s
+  call void @foo(i32 %t10)
+  %t11 = add i32 %b11, %s
+  call void @foo(i32 %t11)
+  %t12 = add i32 %b12, %s
+  call void @foo(i32 %t12)
+  %t13 = add i32 %b13, %s
+  call void @foo(i32 %t13)
+  %t14 = add i32 %b14, %s
+  call void @foo(i32 %t14)
+  %t15 = add i32 %b15, %s
+  call void @foo(i32 %t15)
+  %t16 = add i32 %b16, %s
+  call void @foo(i32 %t16)
+  %t17 = add i32 %b17, %s
+  call void @foo(i32 %t17)
+  %u1 = add i32 %b1, %s2
+  call void @foo(i32 %u1)
+  %u2 = add i32 %b2, %s2
+  call void @foo(i32 %u2)
+  %u3 = add i32 %b3, %s2
+  call void @foo(i32 %u3)
+  %u4 = add i32 %b4, %s2
+  call void @foo(i32 %u4)
+  %u5 = add i32 %b5, %s2
+  call void @foo(i32 %u5)
+  %u6 = add i32 %b6, %s2
+  call void @foo(i32 %u6)
+  %u7 = add i32 %b7, %s2
+  call void @foo(i32 %u7)
+  %u8 = add i32 %b8, %s2
+  call void @foo(i32 %u8)
+  %u9 = add i32 %b9, %s2
+  call void @foo(i32 %u9)
+  %u10 = add i32 %b10, %s2
+  call void @foo(i32 %u10)
+  %u11 = add i32 %b11, %s2
+  call void @foo(i32 %u11)
+  %u12 = add i32 %b12, %s2
+  call void @foo(i32 %u12)
+  %u13 = add i32 %b13, %s2
+  call void @foo(i32 %u13)
+  %u14 = add i32 %b14, %s2
+  call void @foo(i32 %u14)
+  %u15 = add i32 %b15, %s2
+  call void @foo(i32 %u15)
+  %u16 = add i32 %b16, %s2
+  call void @foo(i32 %u16)
+  %u17 = add i32 %b17, %s2
+  call void @foo(i32 %u17)
+  call void @foo(i32 %b1)
+  call void @foo(i32 %b2)
+  call void @foo(i32 %b3)
+  call void @foo(i32 %b4)
+  call void @foo(i32 %b5)
+  call void @foo(i32 %b6)
+  call void @foo(i32 %b7)
+  call void @foo(i32 %b8)
+  call void @foo(i32 %b9)
+  call void @foo(i32 %b10)
+  call void @foo(i32 %b11)
+  call void @foo(i32 %b12)
+  call void @foo(i32 %b13)
+  call void @foo(i32 %b14)
+  call void @foo(i32 %b15)
+  call void @foo(i32 %b16)
+  call void @foo(i32 %b17)
+  ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK: {{.*}}

>From 888f93e45cd34e5ccc15390d3febccf6d9a4795b Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Wed, 2 Sep 2026 23:34:06 +0000
Subject: [PATCH 08/13] Remove trivial knobs, shorten tests

---
 .../AMDGPU/AMDGPUTargetTransformInfo.cpp      |    9 +-
 .../Scalar/StraightLineStrengthReduce.cpp     |   51 +-
 .../AMDGPU/slsr-rp-filter.ll                  | 1888 ++---------------
 .../slsr-rp-filter.ll                         |  384 +---
 4 files changed, 298 insertions(+), 2034 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 35a0d63e77f54..16555675edda3 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -311,13 +311,8 @@ unsigned GCNTTIImpl::getNumberOfRegisters(unsigned RCID) const {
 
 std::optional<unsigned> GCNTTIImpl::getRegisterBudget(const Function &F) const {
   // Report the VGPR budget implied by the occupancy F is compiled for. Callers
-  // comparing a single lumped pressure number against this are conservative on
-  // the SGPR side, which is intentional: whether a value lands in an SGPR or a
-  // VGPR depends on divergence, not on its type.
-  //
-  // On GFX90A this is the combined VGPR+AGPR budget; see getMaxNumVectorRegs
-  // for the split. In dynamic VGPR mode "amdgpu-waves-per-eu" implies no VGPR
-  // limit at all, so this degrades to the full register file.
+  // comparing a single lumped pressure number against this should be
+  // conservative on the SGPR side, which is intentional.
   return ST->getMaxNumVGPRs(F);
 }
 
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index ae39823827087..45dc1942e1bb7 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -125,32 +125,11 @@ static cl::opt<bool>
     EnablePoisonReuseGuard("enable-poison-reuse-guard", cl::init(true),
                            cl::desc("Enable poison-reuse guard"));
 
-// RPFilter targets one pathological shape: a block holding many distinct
-// bases. Each basis contributes one extended live range no matter how many
-// candidates are rewritten against it, so it is the number of distinct bases,
-// not the number of candidates, that tracks how many new concurrent live
-// ranges SLSR would create. Below this count no block in the function can
-// exhibit the pathology, and the liveness and pressure analyses are skipped.
-static constexpr unsigned MinDistinctBasesToFilter = 16;
-
 static cl::opt<bool> EnableRPFilter(
     "slsr-rp-filter", cl::init(true), cl::Hidden,
     cl::desc("SLSR: skip rewrites in blocks where they would push register "
              "pressure past the target's register budget"));
 
-static cl::opt<unsigned> SLSRRegBudget(
-    "slsr-reg-budget", cl::init(0), cl::Hidden,
-    cl::desc("SLSR: override the register budget reported by TTI"));
-
-static cl::opt<double> SLSRRPSafeFraction(
-    "slsr-rp-safe-fraction", cl::init(0.9), cl::Hidden,
-    cl::desc("SLSR: fraction of the register budget treated as safe"));
-
-static cl::opt<unsigned> SLSRRPAbsDelta(
-    "slsr-rp-abs-delta", cl::init(4), cl::Hidden,
-    cl::desc("SLSR: pressure increase tolerated in a block that is already "
-             "over the register budget"));
-
 STATISTIC(NumSCEVCandidateBasisDifferences,
           "Number of candidate-basis SCEV differences computed by SLSR");
 STATISTIC(NumRPFilteredBlocks,
@@ -1447,10 +1426,15 @@ bool StraightLineStrengthReduceLegacyPass::runOnFunction(Function &F) {
 
 namespace {
 
-// TODO: Currently, I am considering (Basis, Cand) pair that are both in the
-// same BB.
-//       The restriction may not be needed.
 class RPFilter {
+  // RPFilter targets one pathological shape: a block holding many distinct
+  // bases. Each basis contributes one extended live range no matter how many
+  // candidates are rewritten against it, so it is the number of distinct bases,
+  // not the number of candidates, that tracks how many new concurrent live
+  // ranges SLSR would create. Below this count no block in the function can
+  // exhibit the pathology, and the liveness and pressure analyses are skipped.
+  static constexpr unsigned MinDistinctBasesToFilter = 16;
+
 public:
   using Candidate = StraightLineStrengthReduce::Candidate;
 
@@ -1535,8 +1519,6 @@ class RPFilter {
   }
 
   std::optional<unsigned> getRegisterBudget() const {
-    if (SLSRRegBudget > 0)
-      return SLSRRegBudget.getValue();
     std::optional<unsigned> Budget = TTI->getRegisterBudget(*F);
     if (Budget && *Budget == 0)
       return std::nullopt;
@@ -1550,6 +1532,7 @@ class RPFilter {
                                   unsigned Budget) const {
     // Leave the allocator some slack: it also has to satisfy register class
     // and ABI constraints that this estimate knows nothing about.
+    constexpr double SLSRRPSafeFraction = 0.9;
     unsigned SafeBudget = static_cast<unsigned>(Budget * SLSRRPSafeFraction);
 
     // There is headroom, so however much the rewrite adds is irrelevant.
@@ -1562,6 +1545,7 @@ class RPFilter {
 
     // Already over budget. SLSR can still lower pressure here, so only refuse
     // rewrites that make it meaningfully worse.
+    constexpr unsigned SLSRRPAbsDelta = 4;
     return After > Before && After - Before > SLSRRPAbsDelta;
   }
 
@@ -1701,10 +1685,6 @@ class RPFilter {
     });
   }
 
-  bool isDebugBlock(const BasicBlock &BB) const {
-    return BB.getName() == "for.cond.cleanup";
-  }
-
   // Return true if I is a candidate and its basis is in the same bb, false
   // otherwise.
   bool insertBasisIfCand(const Instruction *I, ValueSet &LiveSetWithSLSR,
@@ -1714,6 +1694,11 @@ class RPFilter {
       return false;
 
     // I is a Candidate
+    // We consider here only the case both Cand and Basis are in the same BB.
+    // Therefore, if an original operand of Cand is a liveout, it means it was
+    // used outside the BB, and will stay so. Likewise, if a Basis is a liveout,
+    // it means it was used outside the BB, and will stay so. SLSR does not
+    // delete existing defs or create new defs.
     const Instruction *Basis = It->second->Basis->Ins;
     if (Basis->getParent() == I->getParent()) {
       LiveSetWithSLSR.insert(Basis);
@@ -1746,12 +1731,6 @@ class RPFilter {
     // to
     // Basis = ..
     // Cand = f(Basis, Delta, // possibly some of the original operands ..)
-    //
-    // We consider here only the case both Cand and Basis are in the same BB.
-    // Therefore, if an original operand of Cand is a liveout, it means it was
-    // used outside the BB, and will stay so. Likewise, if a Basis is a liveout,
-    // it means it was used outside the BB, and will stay so. SLSR does not
-    // delete existing defs or create new defs.
     ValueSet LiveSetWithSLSR = LiveOut;
     unsigned MaxWWithSLSR = MaxW;
 
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
index 24ae488d9a7f3..a5f86a1a42ce4 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
@@ -1,7 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt < %s -mtriple=amdgpu9.50-amd-amdhsa -passes=slsr -S -slsr-reg-budget=32 | FileCheck %s --check-prefixes=CHECK,BUDGET32
-; RUN: opt < %s -mtriple=amdgpu9.50-amd-amdhsa -passes=slsr -S | FileCheck %s --check-prefixes=CHECK,ATTRS
-; RUN: opt < %s -mtriple=amdgpu9.50-amd-amdhsa -passes=slsr -S -slsr-rp-filter=false | FileCheck %s --check-prefixes=CHECK,NOFILTER
+; RUN: opt < %s -mtriple=amdgpu-amdhsa-amd -passes=slsr -S | FileCheck %s
 
 ; SLSR rewrites a candidate as its basis plus a delta, which keeps the basis live
 ; down to the candidate. The register-pressure filter drops a block's rewrites
@@ -9,769 +7,32 @@
 ; allocate.
 ;
 ; The budget comes from TTI, which on AMDGPU reports the VGPR count implied by
-; the occupancy the function is compiled for, and -slsr-reg-budget overrides it.
-; -slsr-rp-safe-fraction (0.9 by default) is applied on top, so a budget of 32
-; leaves 28 usable registers. The filter does nothing unless -slsr-rp-filter is
-; passed, and it only examines functions where some block holds more than 16
-; distinct bases.
+; the occupancy the function is compiled for. This triple selects the generic
+; GPU, which has 256 VGPRs in total. Waves per EU follows from the flat work
+; group size, the budget is those 256 registers divided by the waves that must
+; share an EU, and the filter treats 0.9 of the budget as usable.
 ;
-; @many_bases_overlapping has 17 bases and goes from 20 to 36 registers, crossing
-; the 28-register safe budget, so its rewrites are dropped. @many_bases_adjacent
-; has the same 17 bases, but each %u<k> immediately follows its %t<k>, so the
-; extended ranges never coexist and the peak only reaches 21, which fits.
+; Both functions below have byte-identical bodies: 17 distinct bases whose
+; rewrites take the block's peak pressure from 80 registers to 144. The only
+; difference is the occupancy attribute, so the budget is the only variable and
+; the comparison isolates it.
 ;
-; @four_bases holds only 4 bases, below the gate, so the filter never looks at it
-; and its rewrites stand. Its i128 values put the block at 28 registers rising to
-; 44, which would cross the 28-register safe budget if it were examined, so the
-; gate really is the only thing sparing it.
+;   @peak_above_budget  no attribute. The work group size defaults to 1024
+;                       threads, which is 16 wave64s over 4 EUs, so 4 waves per
+;                       EU and a budget of 256/4 = 64 registers, 57 usable.
+;                       144 is above 57, so the rewrites are dropped.
 ;
-; The i32 functions carry no occupancy attributes, so their flat work group size
-; defaults to 1024, which is 16 wave64s spread over 4 EUs, hence 4 waves per EU
-; and a budget of 512/4 = 128 registers. Their peak of 36 fits, so the ATTRS run
-; leaves them alone and only the forced budget of 32 filters them.
-;
-; The i128 functions at the end raise the peak to 144 so the real budget decides
-; without any override. All three have identical bodies and identical pressure,
-; so the attributes are the only variable:
-;
-;   @default_occupancy      no attributes           4 waves/EU  budget 128  filtered
-;   @small_work_group       flat-work-group-size    1 wave/EU   budget 512  kept
-;   @small_wg_high_occ      + waves-per-eu=8        8 waves/EU  budget 64   filtered
-;
-; The first pair shows flat-work-group-size raising the budget, and the second
-; shows waves-per-eu lowering it again. NOFILTER switches the analysis off.
+;   @peak_below_budget  256 threads is 4 wave64s over 4 EUs, so 1 wave per EU
+;                       and a budget of the full 256 registers, 230 usable.
+;                       The same 144 is below 230, so the rewrites survive.
 
-declare void @foo(i32)
 declare void @bar(i128)
 
-define void @many_bases_overlapping(i32 %s, i32 %b1, i32 %b2, i32 %b3, i32 %b4, i32 %b5, i32 %b6, i32 %b7, i32 %b8, i32 %b9, i32 %b10, i32 %b11, i32 %b12, i32 %b13, i32 %b14, i32 %b15, i32 %b16, i32 %b17) {
-; FILTER-LABEL: define void @many_bases_overlapping(
-; FILTER-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; FILTER-NEXT:  [[ENTRY:.*:]]
-; FILTER-NEXT:    [[S2:%.*]] = shl i32 [[S]], 1
-; FILTER-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T1]])
-; FILTER-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T2]])
-; FILTER-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T3]])
-; FILTER-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T4]])
-; FILTER-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T5]])
-; FILTER-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T6]])
-; FILTER-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T7]])
-; FILTER-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T8]])
-; FILTER-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T9]])
-; FILTER-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T10]])
-; FILTER-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T11]])
-; FILTER-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T12]])
-; FILTER-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T13]])
-; FILTER-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T14]])
-; FILTER-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T15]])
-; FILTER-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T16]])
-; FILTER-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
-; FILTER-NEXT:    call void @foo(i32 [[T17]])
-; FILTER-NEXT:    [[U1:%.*]] = add i32 [[B1]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U1]])
-; FILTER-NEXT:    [[U2:%.*]] = add i32 [[B2]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U2]])
-; FILTER-NEXT:    [[U3:%.*]] = add i32 [[B3]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U3]])
-; FILTER-NEXT:    [[U4:%.*]] = add i32 [[B4]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U4]])
-; FILTER-NEXT:    [[U5:%.*]] = add i32 [[B5]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U5]])
-; FILTER-NEXT:    [[U6:%.*]] = add i32 [[B6]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U6]])
-; FILTER-NEXT:    [[U7:%.*]] = add i32 [[B7]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U7]])
-; FILTER-NEXT:    [[U8:%.*]] = add i32 [[B8]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U8]])
-; FILTER-NEXT:    [[U9:%.*]] = add i32 [[B9]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U9]])
-; FILTER-NEXT:    [[U10:%.*]] = add i32 [[B10]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U10]])
-; FILTER-NEXT:    [[U11:%.*]] = add i32 [[B11]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U11]])
-; FILTER-NEXT:    [[U12:%.*]] = add i32 [[B12]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U12]])
-; FILTER-NEXT:    [[U13:%.*]] = add i32 [[B13]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U13]])
-; FILTER-NEXT:    [[U14:%.*]] = add i32 [[B14]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U14]])
-; FILTER-NEXT:    [[U15:%.*]] = add i32 [[B15]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U15]])
-; FILTER-NEXT:    [[U16:%.*]] = add i32 [[B16]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U16]])
-; FILTER-NEXT:    [[U17:%.*]] = add i32 [[B17]], [[S2]]
-; FILTER-NEXT:    call void @foo(i32 [[U17]])
-; FILTER-NEXT:    call void @foo(i32 [[B1]])
-; FILTER-NEXT:    call void @foo(i32 [[B2]])
-; FILTER-NEXT:    call void @foo(i32 [[B3]])
-; FILTER-NEXT:    call void @foo(i32 [[B4]])
-; FILTER-NEXT:    call void @foo(i32 [[B5]])
-; FILTER-NEXT:    call void @foo(i32 [[B6]])
-; FILTER-NEXT:    call void @foo(i32 [[B7]])
-; FILTER-NEXT:    call void @foo(i32 [[B8]])
-; FILTER-NEXT:    call void @foo(i32 [[B9]])
-; FILTER-NEXT:    call void @foo(i32 [[B10]])
-; FILTER-NEXT:    call void @foo(i32 [[B11]])
-; FILTER-NEXT:    call void @foo(i32 [[B12]])
-; FILTER-NEXT:    call void @foo(i32 [[B13]])
-; FILTER-NEXT:    call void @foo(i32 [[B14]])
-; FILTER-NEXT:    call void @foo(i32 [[B15]])
-; FILTER-NEXT:    call void @foo(i32 [[B16]])
-; FILTER-NEXT:    call void @foo(i32 [[B17]])
-; FILTER-NEXT:    ret void
-;
-; OFF-LABEL: define void @many_bases_overlapping(
-; OFF-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; OFF-NEXT:  [[ENTRY:.*:]]
-; OFF-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T1]])
-; OFF-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T2]])
-; OFF-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T3]])
-; OFF-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T4]])
-; OFF-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T5]])
-; OFF-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T6]])
-; OFF-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T7]])
-; OFF-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T8]])
-; OFF-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T9]])
-; OFF-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T10]])
-; OFF-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T11]])
-; OFF-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T12]])
-; OFF-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T13]])
-; OFF-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T14]])
-; OFF-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T15]])
-; OFF-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T16]])
-; OFF-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[T17]])
-; OFF-NEXT:    [[U1:%.*]] = add i32 [[T1]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U1]])
-; OFF-NEXT:    [[U2:%.*]] = add i32 [[T2]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U2]])
-; OFF-NEXT:    [[U3:%.*]] = add i32 [[T3]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U3]])
-; OFF-NEXT:    [[U4:%.*]] = add i32 [[T4]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U4]])
-; OFF-NEXT:    [[U5:%.*]] = add i32 [[T5]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U5]])
-; OFF-NEXT:    [[U6:%.*]] = add i32 [[T6]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U6]])
-; OFF-NEXT:    [[U7:%.*]] = add i32 [[T7]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U7]])
-; OFF-NEXT:    [[U8:%.*]] = add i32 [[T8]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U8]])
-; OFF-NEXT:    [[U9:%.*]] = add i32 [[T9]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U9]])
-; OFF-NEXT:    [[U10:%.*]] = add i32 [[T10]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U10]])
-; OFF-NEXT:    [[U11:%.*]] = add i32 [[T11]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U11]])
-; OFF-NEXT:    [[U12:%.*]] = add i32 [[T12]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U12]])
-; OFF-NEXT:    [[U13:%.*]] = add i32 [[T13]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U13]])
-; OFF-NEXT:    [[U14:%.*]] = add i32 [[T14]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U14]])
-; OFF-NEXT:    [[U15:%.*]] = add i32 [[T15]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U15]])
-; OFF-NEXT:    [[U16:%.*]] = add i32 [[T16]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U16]])
-; OFF-NEXT:    [[U17:%.*]] = add i32 [[T17]], [[S]]
-; OFF-NEXT:    call void @foo(i32 [[U17]])
-; OFF-NEXT:    call void @foo(i32 [[B1]])
-; OFF-NEXT:    call void @foo(i32 [[B2]])
-; OFF-NEXT:    call void @foo(i32 [[B3]])
-; OFF-NEXT:    call void @foo(i32 [[B4]])
-; OFF-NEXT:    call void @foo(i32 [[B5]])
-; OFF-NEXT:    call void @foo(i32 [[B6]])
-; OFF-NEXT:    call void @foo(i32 [[B7]])
-; OFF-NEXT:    call void @foo(i32 [[B8]])
-; OFF-NEXT:    call void @foo(i32 [[B9]])
-; OFF-NEXT:    call void @foo(i32 [[B10]])
-; OFF-NEXT:    call void @foo(i32 [[B11]])
-; OFF-NEXT:    call void @foo(i32 [[B12]])
-; OFF-NEXT:    call void @foo(i32 [[B13]])
-; OFF-NEXT:    call void @foo(i32 [[B14]])
-; OFF-NEXT:    call void @foo(i32 [[B15]])
-; OFF-NEXT:    call void @foo(i32 [[B16]])
-; OFF-NEXT:    call void @foo(i32 [[B17]])
-; OFF-NEXT:    ret void
-;
-; BUDGET32-LABEL: define void @many_bases_overlapping(
-; BUDGET32-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; BUDGET32-NEXT:  [[ENTRY:.*:]]
-; BUDGET32-NEXT:    [[S2:%.*]] = shl i32 [[S]], 1
-; BUDGET32-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T1]])
-; BUDGET32-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T2]])
-; BUDGET32-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T3]])
-; BUDGET32-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T4]])
-; BUDGET32-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T5]])
-; BUDGET32-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T6]])
-; BUDGET32-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T7]])
-; BUDGET32-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T8]])
-; BUDGET32-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T9]])
-; BUDGET32-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T10]])
-; BUDGET32-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T11]])
-; BUDGET32-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T12]])
-; BUDGET32-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T13]])
-; BUDGET32-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T14]])
-; BUDGET32-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T15]])
-; BUDGET32-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T16]])
-; BUDGET32-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
-; BUDGET32-NEXT:    call void @foo(i32 [[T17]])
-; BUDGET32-NEXT:    [[U1:%.*]] = add i32 [[B1]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U1]])
-; BUDGET32-NEXT:    [[U2:%.*]] = add i32 [[B2]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U2]])
-; BUDGET32-NEXT:    [[U3:%.*]] = add i32 [[B3]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U3]])
-; BUDGET32-NEXT:    [[U4:%.*]] = add i32 [[B4]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U4]])
-; BUDGET32-NEXT:    [[U5:%.*]] = add i32 [[B5]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U5]])
-; BUDGET32-NEXT:    [[U6:%.*]] = add i32 [[B6]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U6]])
-; BUDGET32-NEXT:    [[U7:%.*]] = add i32 [[B7]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U7]])
-; BUDGET32-NEXT:    [[U8:%.*]] = add i32 [[B8]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U8]])
-; BUDGET32-NEXT:    [[U9:%.*]] = add i32 [[B9]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U9]])
-; BUDGET32-NEXT:    [[U10:%.*]] = add i32 [[B10]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U10]])
-; BUDGET32-NEXT:    [[U11:%.*]] = add i32 [[B11]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U11]])
-; BUDGET32-NEXT:    [[U12:%.*]] = add i32 [[B12]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U12]])
-; BUDGET32-NEXT:    [[U13:%.*]] = add i32 [[B13]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U13]])
-; BUDGET32-NEXT:    [[U14:%.*]] = add i32 [[B14]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U14]])
-; BUDGET32-NEXT:    [[U15:%.*]] = add i32 [[B15]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U15]])
-; BUDGET32-NEXT:    [[U16:%.*]] = add i32 [[B16]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U16]])
-; BUDGET32-NEXT:    [[U17:%.*]] = add i32 [[B17]], [[S2]]
-; BUDGET32-NEXT:    call void @foo(i32 [[U17]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B1]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B2]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B3]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B4]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B5]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B6]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B7]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B8]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B9]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B10]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B11]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B12]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B13]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B14]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B15]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B16]])
-; BUDGET32-NEXT:    call void @foo(i32 [[B17]])
-; BUDGET32-NEXT:    ret void
-;
-; ATTRS-LABEL: define void @many_bases_overlapping(
-; ATTRS-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; ATTRS-NEXT:  [[ENTRY:.*:]]
-; ATTRS-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T1]])
-; ATTRS-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T2]])
-; ATTRS-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T3]])
-; ATTRS-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T4]])
-; ATTRS-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T5]])
-; ATTRS-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T6]])
-; ATTRS-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T7]])
-; ATTRS-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T8]])
-; ATTRS-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T9]])
-; ATTRS-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T10]])
-; ATTRS-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T11]])
-; ATTRS-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T12]])
-; ATTRS-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T13]])
-; ATTRS-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T14]])
-; ATTRS-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T15]])
-; ATTRS-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T16]])
-; ATTRS-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[T17]])
-; ATTRS-NEXT:    [[U1:%.*]] = add i32 [[T1]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U1]])
-; ATTRS-NEXT:    [[U2:%.*]] = add i32 [[T2]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U2]])
-; ATTRS-NEXT:    [[U3:%.*]] = add i32 [[T3]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U3]])
-; ATTRS-NEXT:    [[U4:%.*]] = add i32 [[T4]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U4]])
-; ATTRS-NEXT:    [[U5:%.*]] = add i32 [[T5]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U5]])
-; ATTRS-NEXT:    [[U6:%.*]] = add i32 [[T6]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U6]])
-; ATTRS-NEXT:    [[U7:%.*]] = add i32 [[T7]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U7]])
-; ATTRS-NEXT:    [[U8:%.*]] = add i32 [[T8]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U8]])
-; ATTRS-NEXT:    [[U9:%.*]] = add i32 [[T9]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U9]])
-; ATTRS-NEXT:    [[U10:%.*]] = add i32 [[T10]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U10]])
-; ATTRS-NEXT:    [[U11:%.*]] = add i32 [[T11]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U11]])
-; ATTRS-NEXT:    [[U12:%.*]] = add i32 [[T12]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U12]])
-; ATTRS-NEXT:    [[U13:%.*]] = add i32 [[T13]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U13]])
-; ATTRS-NEXT:    [[U14:%.*]] = add i32 [[T14]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U14]])
-; ATTRS-NEXT:    [[U15:%.*]] = add i32 [[T15]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U15]])
-; ATTRS-NEXT:    [[U16:%.*]] = add i32 [[T16]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U16]])
-; ATTRS-NEXT:    [[U17:%.*]] = add i32 [[T17]], [[S]]
-; ATTRS-NEXT:    call void @foo(i32 [[U17]])
-; ATTRS-NEXT:    call void @foo(i32 [[B1]])
-; ATTRS-NEXT:    call void @foo(i32 [[B2]])
-; ATTRS-NEXT:    call void @foo(i32 [[B3]])
-; ATTRS-NEXT:    call void @foo(i32 [[B4]])
-; ATTRS-NEXT:    call void @foo(i32 [[B5]])
-; ATTRS-NEXT:    call void @foo(i32 [[B6]])
-; ATTRS-NEXT:    call void @foo(i32 [[B7]])
-; ATTRS-NEXT:    call void @foo(i32 [[B8]])
-; ATTRS-NEXT:    call void @foo(i32 [[B9]])
-; ATTRS-NEXT:    call void @foo(i32 [[B10]])
-; ATTRS-NEXT:    call void @foo(i32 [[B11]])
-; ATTRS-NEXT:    call void @foo(i32 [[B12]])
-; ATTRS-NEXT:    call void @foo(i32 [[B13]])
-; ATTRS-NEXT:    call void @foo(i32 [[B14]])
-; ATTRS-NEXT:    call void @foo(i32 [[B15]])
-; ATTRS-NEXT:    call void @foo(i32 [[B16]])
-; ATTRS-NEXT:    call void @foo(i32 [[B17]])
-; ATTRS-NEXT:    ret void
-;
-; NOFILTER-LABEL: define void @many_bases_overlapping(
-; NOFILTER-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; NOFILTER-NEXT:  [[ENTRY:.*:]]
-; NOFILTER-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T1]])
-; NOFILTER-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T2]])
-; NOFILTER-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T3]])
-; NOFILTER-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T4]])
-; NOFILTER-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T5]])
-; NOFILTER-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T6]])
-; NOFILTER-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T7]])
-; NOFILTER-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T8]])
-; NOFILTER-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T9]])
-; NOFILTER-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T10]])
-; NOFILTER-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T11]])
-; NOFILTER-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T12]])
-; NOFILTER-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T13]])
-; NOFILTER-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T14]])
-; NOFILTER-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T15]])
-; NOFILTER-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T16]])
-; NOFILTER-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[T17]])
-; NOFILTER-NEXT:    [[U1:%.*]] = add i32 [[T1]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U1]])
-; NOFILTER-NEXT:    [[U2:%.*]] = add i32 [[T2]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U2]])
-; NOFILTER-NEXT:    [[U3:%.*]] = add i32 [[T3]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U3]])
-; NOFILTER-NEXT:    [[U4:%.*]] = add i32 [[T4]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U4]])
-; NOFILTER-NEXT:    [[U5:%.*]] = add i32 [[T5]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U5]])
-; NOFILTER-NEXT:    [[U6:%.*]] = add i32 [[T6]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U6]])
-; NOFILTER-NEXT:    [[U7:%.*]] = add i32 [[T7]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U7]])
-; NOFILTER-NEXT:    [[U8:%.*]] = add i32 [[T8]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U8]])
-; NOFILTER-NEXT:    [[U9:%.*]] = add i32 [[T9]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U9]])
-; NOFILTER-NEXT:    [[U10:%.*]] = add i32 [[T10]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U10]])
-; NOFILTER-NEXT:    [[U11:%.*]] = add i32 [[T11]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U11]])
-; NOFILTER-NEXT:    [[U12:%.*]] = add i32 [[T12]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U12]])
-; NOFILTER-NEXT:    [[U13:%.*]] = add i32 [[T13]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U13]])
-; NOFILTER-NEXT:    [[U14:%.*]] = add i32 [[T14]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U14]])
-; NOFILTER-NEXT:    [[U15:%.*]] = add i32 [[T15]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U15]])
-; NOFILTER-NEXT:    [[U16:%.*]] = add i32 [[T16]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U16]])
-; NOFILTER-NEXT:    [[U17:%.*]] = add i32 [[T17]], [[S]]
-; NOFILTER-NEXT:    call void @foo(i32 [[U17]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B1]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B2]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B3]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B4]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B5]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B6]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B7]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B8]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B9]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B10]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B11]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B12]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B13]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B14]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B15]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B16]])
-; NOFILTER-NEXT:    call void @foo(i32 [[B17]])
-; NOFILTER-NEXT:    ret void
-;
-entry:
-  %s2 = shl i32 %s, 1
-  %t1 = add i32 %b1, %s
-  call void @foo(i32 %t1)
-  %t2 = add i32 %b2, %s
-  call void @foo(i32 %t2)
-  %t3 = add i32 %b3, %s
-  call void @foo(i32 %t3)
-  %t4 = add i32 %b4, %s
-  call void @foo(i32 %t4)
-  %t5 = add i32 %b5, %s
-  call void @foo(i32 %t5)
-  %t6 = add i32 %b6, %s
-  call void @foo(i32 %t6)
-  %t7 = add i32 %b7, %s
-  call void @foo(i32 %t7)
-  %t8 = add i32 %b8, %s
-  call void @foo(i32 %t8)
-  %t9 = add i32 %b9, %s
-  call void @foo(i32 %t9)
-  %t10 = add i32 %b10, %s
-  call void @foo(i32 %t10)
-  %t11 = add i32 %b11, %s
-  call void @foo(i32 %t11)
-  %t12 = add i32 %b12, %s
-  call void @foo(i32 %t12)
-  %t13 = add i32 %b13, %s
-  call void @foo(i32 %t13)
-  %t14 = add i32 %b14, %s
-  call void @foo(i32 %t14)
-  %t15 = add i32 %b15, %s
-  call void @foo(i32 %t15)
-  %t16 = add i32 %b16, %s
-  call void @foo(i32 %t16)
-  %t17 = add i32 %b17, %s
-  call void @foo(i32 %t17)
-  %u1 = add i32 %b1, %s2
-  call void @foo(i32 %u1)
-  %u2 = add i32 %b2, %s2
-  call void @foo(i32 %u2)
-  %u3 = add i32 %b3, %s2
-  call void @foo(i32 %u3)
-  %u4 = add i32 %b4, %s2
-  call void @foo(i32 %u4)
-  %u5 = add i32 %b5, %s2
-  call void @foo(i32 %u5)
-  %u6 = add i32 %b6, %s2
-  call void @foo(i32 %u6)
-  %u7 = add i32 %b7, %s2
-  call void @foo(i32 %u7)
-  %u8 = add i32 %b8, %s2
-  call void @foo(i32 %u8)
-  %u9 = add i32 %b9, %s2
-  call void @foo(i32 %u9)
-  %u10 = add i32 %b10, %s2
-  call void @foo(i32 %u10)
-  %u11 = add i32 %b11, %s2
-  call void @foo(i32 %u11)
-  %u12 = add i32 %b12, %s2
-  call void @foo(i32 %u12)
-  %u13 = add i32 %b13, %s2
-  call void @foo(i32 %u13)
-  %u14 = add i32 %b14, %s2
-  call void @foo(i32 %u14)
-  %u15 = add i32 %b15, %s2
-  call void @foo(i32 %u15)
-  %u16 = add i32 %b16, %s2
-  call void @foo(i32 %u16)
-  %u17 = add i32 %b17, %s2
-  call void @foo(i32 %u17)
-  call void @foo(i32 %b1)
-  call void @foo(i32 %b2)
-  call void @foo(i32 %b3)
-  call void @foo(i32 %b4)
-  call void @foo(i32 %b5)
-  call void @foo(i32 %b6)
-  call void @foo(i32 %b7)
-  call void @foo(i32 %b8)
-  call void @foo(i32 %b9)
-  call void @foo(i32 %b10)
-  call void @foo(i32 %b11)
-  call void @foo(i32 %b12)
-  call void @foo(i32 %b13)
-  call void @foo(i32 %b14)
-  call void @foo(i32 %b15)
-  call void @foo(i32 %b16)
-  call void @foo(i32 %b17)
-  ret void
-}
-
-define void @many_bases_adjacent(i32 %s, i32 %b1, i32 %b2, i32 %b3, i32 %b4, i32 %b5, i32 %b6, i32 %b7, i32 %b8, i32 %b9, i32 %b10, i32 %b11, i32 %b12, i32 %b13, i32 %b14, i32 %b15, i32 %b16, i32 %b17) {
-; CHECK-LABEL: define void @many_bases_adjacent(
-; CHECK-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T1]])
-; CHECK-NEXT:    [[U1:%.*]] = add i32 [[T1]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U1]])
-; CHECK-NEXT:    call void @foo(i32 [[B1]])
-; CHECK-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T2]])
-; CHECK-NEXT:    [[U2:%.*]] = add i32 [[T2]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U2]])
-; CHECK-NEXT:    call void @foo(i32 [[B2]])
-; CHECK-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T3]])
-; CHECK-NEXT:    [[U3:%.*]] = add i32 [[T3]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U3]])
-; CHECK-NEXT:    call void @foo(i32 [[B3]])
-; CHECK-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T4]])
-; CHECK-NEXT:    [[U4:%.*]] = add i32 [[T4]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U4]])
-; CHECK-NEXT:    call void @foo(i32 [[B4]])
-; CHECK-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T5]])
-; CHECK-NEXT:    [[U5:%.*]] = add i32 [[T5]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U5]])
-; CHECK-NEXT:    call void @foo(i32 [[B5]])
-; CHECK-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T6]])
-; CHECK-NEXT:    [[U6:%.*]] = add i32 [[T6]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U6]])
-; CHECK-NEXT:    call void @foo(i32 [[B6]])
-; CHECK-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T7]])
-; CHECK-NEXT:    [[U7:%.*]] = add i32 [[T7]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U7]])
-; CHECK-NEXT:    call void @foo(i32 [[B7]])
-; CHECK-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T8]])
-; CHECK-NEXT:    [[U8:%.*]] = add i32 [[T8]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U8]])
-; CHECK-NEXT:    call void @foo(i32 [[B8]])
-; CHECK-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T9]])
-; CHECK-NEXT:    [[U9:%.*]] = add i32 [[T9]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U9]])
-; CHECK-NEXT:    call void @foo(i32 [[B9]])
-; CHECK-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T10]])
-; CHECK-NEXT:    [[U10:%.*]] = add i32 [[T10]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U10]])
-; CHECK-NEXT:    call void @foo(i32 [[B10]])
-; CHECK-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T11]])
-; CHECK-NEXT:    [[U11:%.*]] = add i32 [[T11]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U11]])
-; CHECK-NEXT:    call void @foo(i32 [[B11]])
-; CHECK-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T12]])
-; CHECK-NEXT:    [[U12:%.*]] = add i32 [[T12]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U12]])
-; CHECK-NEXT:    call void @foo(i32 [[B12]])
-; CHECK-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T13]])
-; CHECK-NEXT:    [[U13:%.*]] = add i32 [[T13]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U13]])
-; CHECK-NEXT:    call void @foo(i32 [[B13]])
-; CHECK-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T14]])
-; CHECK-NEXT:    [[U14:%.*]] = add i32 [[T14]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U14]])
-; CHECK-NEXT:    call void @foo(i32 [[B14]])
-; CHECK-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T15]])
-; CHECK-NEXT:    [[U15:%.*]] = add i32 [[T15]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U15]])
-; CHECK-NEXT:    call void @foo(i32 [[B15]])
-; CHECK-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T16]])
-; CHECK-NEXT:    [[U16:%.*]] = add i32 [[T16]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U16]])
-; CHECK-NEXT:    call void @foo(i32 [[B16]])
-; CHECK-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[T17]])
-; CHECK-NEXT:    [[U17:%.*]] = add i32 [[T17]], [[S]]
-; CHECK-NEXT:    call void @foo(i32 [[U17]])
-; CHECK-NEXT:    call void @foo(i32 [[B17]])
-; CHECK-NEXT:    ret void
-;
-entry:
-  %s2 = shl i32 %s, 1
-  %t1 = add i32 %b1, %s
-  call void @foo(i32 %t1)
-  %u1 = add i32 %b1, %s2
-  call void @foo(i32 %u1)
-  call void @foo(i32 %b1)
-  %t2 = add i32 %b2, %s
-  call void @foo(i32 %t2)
-  %u2 = add i32 %b2, %s2
-  call void @foo(i32 %u2)
-  call void @foo(i32 %b2)
-  %t3 = add i32 %b3, %s
-  call void @foo(i32 %t3)
-  %u3 = add i32 %b3, %s2
-  call void @foo(i32 %u3)
-  call void @foo(i32 %b3)
-  %t4 = add i32 %b4, %s
-  call void @foo(i32 %t4)
-  %u4 = add i32 %b4, %s2
-  call void @foo(i32 %u4)
-  call void @foo(i32 %b4)
-  %t5 = add i32 %b5, %s
-  call void @foo(i32 %t5)
-  %u5 = add i32 %b5, %s2
-  call void @foo(i32 %u5)
-  call void @foo(i32 %b5)
-  %t6 = add i32 %b6, %s
-  call void @foo(i32 %t6)
-  %u6 = add i32 %b6, %s2
-  call void @foo(i32 %u6)
-  call void @foo(i32 %b6)
-  %t7 = add i32 %b7, %s
-  call void @foo(i32 %t7)
-  %u7 = add i32 %b7, %s2
-  call void @foo(i32 %u7)
-  call void @foo(i32 %b7)
-  %t8 = add i32 %b8, %s
-  call void @foo(i32 %t8)
-  %u8 = add i32 %b8, %s2
-  call void @foo(i32 %u8)
-  call void @foo(i32 %b8)
-  %t9 = add i32 %b9, %s
-  call void @foo(i32 %t9)
-  %u9 = add i32 %b9, %s2
-  call void @foo(i32 %u9)
-  call void @foo(i32 %b9)
-  %t10 = add i32 %b10, %s
-  call void @foo(i32 %t10)
-  %u10 = add i32 %b10, %s2
-  call void @foo(i32 %u10)
-  call void @foo(i32 %b10)
-  %t11 = add i32 %b11, %s
-  call void @foo(i32 %t11)
-  %u11 = add i32 %b11, %s2
-  call void @foo(i32 %u11)
-  call void @foo(i32 %b11)
-  %t12 = add i32 %b12, %s
-  call void @foo(i32 %t12)
-  %u12 = add i32 %b12, %s2
-  call void @foo(i32 %u12)
-  call void @foo(i32 %b12)
-  %t13 = add i32 %b13, %s
-  call void @foo(i32 %t13)
-  %u13 = add i32 %b13, %s2
-  call void @foo(i32 %u13)
-  call void @foo(i32 %b13)
-  %t14 = add i32 %b14, %s
-  call void @foo(i32 %t14)
-  %u14 = add i32 %b14, %s2
-  call void @foo(i32 %u14)
-  call void @foo(i32 %b14)
-  %t15 = add i32 %b15, %s
-  call void @foo(i32 %t15)
-  %u15 = add i32 %b15, %s2
-  call void @foo(i32 %u15)
-  call void @foo(i32 %b15)
-  %t16 = add i32 %b16, %s
-  call void @foo(i32 %t16)
-  %u16 = add i32 %b16, %s2
-  call void @foo(i32 %u16)
-  call void @foo(i32 %b16)
-  %t17 = add i32 %b17, %s
-  call void @foo(i32 %t17)
-  %u17 = add i32 %b17, %s2
-  call void @foo(i32 %u17)
-  call void @foo(i32 %b17)
-  ret void
-}
-
-define void @four_bases(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4) {
-; CHECK-LABEL: define void @four_bases(
-; CHECK-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]]) {
+define void @peak_above_budget(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) {
+; CHECK-LABEL: define void @peak_above_budget(
+; CHECK-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[S2:%.*]] = shl i128 [[S]], 1
 ; CHECK-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
 ; CHECK-NEXT:    call void @bar(i128 [[T1]])
 ; CHECK-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
@@ -780,324 +41,85 @@ define void @four_bases(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4) {
 ; CHECK-NEXT:    call void @bar(i128 [[T3]])
 ; CHECK-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
 ; CHECK-NEXT:    call void @bar(i128 [[T4]])
-; CHECK-NEXT:    [[U1:%.*]] = add i128 [[T1]], [[S]]
+; CHECK-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T5]])
+; CHECK-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T6]])
+; CHECK-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T7]])
+; CHECK-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T8]])
+; CHECK-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T9]])
+; CHECK-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T10]])
+; CHECK-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T11]])
+; CHECK-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T12]])
+; CHECK-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T13]])
+; CHECK-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T14]])
+; CHECK-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T15]])
+; CHECK-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T16]])
+; CHECK-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T17]])
+; CHECK-NEXT:    [[U1:%.*]] = add i128 [[B1]], [[S2]]
 ; CHECK-NEXT:    call void @bar(i128 [[U1]])
-; CHECK-NEXT:    [[U2:%.*]] = add i128 [[T2]], [[S]]
+; CHECK-NEXT:    [[U2:%.*]] = add i128 [[B2]], [[S2]]
 ; CHECK-NEXT:    call void @bar(i128 [[U2]])
-; CHECK-NEXT:    [[U3:%.*]] = add i128 [[T3]], [[S]]
+; CHECK-NEXT:    [[U3:%.*]] = add i128 [[B3]], [[S2]]
 ; CHECK-NEXT:    call void @bar(i128 [[U3]])
-; CHECK-NEXT:    [[U4:%.*]] = add i128 [[T4]], [[S]]
+; CHECK-NEXT:    [[U4:%.*]] = add i128 [[B4]], [[S2]]
 ; CHECK-NEXT:    call void @bar(i128 [[U4]])
+; CHECK-NEXT:    [[U5:%.*]] = add i128 [[B5]], [[S2]]
+; CHECK-NEXT:    call void @bar(i128 [[U5]])
+; CHECK-NEXT:    [[U6:%.*]] = add i128 [[B6]], [[S2]]
+; CHECK-NEXT:    call void @bar(i128 [[U6]])
+; CHECK-NEXT:    [[U7:%.*]] = add i128 [[B7]], [[S2]]
+; CHECK-NEXT:    call void @bar(i128 [[U7]])
+; CHECK-NEXT:    [[U8:%.*]] = add i128 [[B8]], [[S2]]
+; CHECK-NEXT:    call void @bar(i128 [[U8]])
+; CHECK-NEXT:    [[U9:%.*]] = add i128 [[B9]], [[S2]]
+; CHECK-NEXT:    call void @bar(i128 [[U9]])
+; CHECK-NEXT:    [[U10:%.*]] = add i128 [[B10]], [[S2]]
+; CHECK-NEXT:    call void @bar(i128 [[U10]])
+; CHECK-NEXT:    [[U11:%.*]] = add i128 [[B11]], [[S2]]
+; CHECK-NEXT:    call void @bar(i128 [[U11]])
+; CHECK-NEXT:    [[U12:%.*]] = add i128 [[B12]], [[S2]]
+; CHECK-NEXT:    call void @bar(i128 [[U12]])
+; CHECK-NEXT:    [[U13:%.*]] = add i128 [[B13]], [[S2]]
+; CHECK-NEXT:    call void @bar(i128 [[U13]])
+; CHECK-NEXT:    [[U14:%.*]] = add i128 [[B14]], [[S2]]
+; CHECK-NEXT:    call void @bar(i128 [[U14]])
+; CHECK-NEXT:    [[U15:%.*]] = add i128 [[B15]], [[S2]]
+; CHECK-NEXT:    call void @bar(i128 [[U15]])
+; CHECK-NEXT:    [[U16:%.*]] = add i128 [[B16]], [[S2]]
+; CHECK-NEXT:    call void @bar(i128 [[U16]])
+; CHECK-NEXT:    [[U17:%.*]] = add i128 [[B17]], [[S2]]
+; CHECK-NEXT:    call void @bar(i128 [[U17]])
 ; CHECK-NEXT:    call void @bar(i128 [[B1]])
 ; CHECK-NEXT:    call void @bar(i128 [[B2]])
 ; CHECK-NEXT:    call void @bar(i128 [[B3]])
 ; CHECK-NEXT:    call void @bar(i128 [[B4]])
+; CHECK-NEXT:    call void @bar(i128 [[B5]])
+; CHECK-NEXT:    call void @bar(i128 [[B6]])
+; CHECK-NEXT:    call void @bar(i128 [[B7]])
+; CHECK-NEXT:    call void @bar(i128 [[B8]])
+; CHECK-NEXT:    call void @bar(i128 [[B9]])
+; CHECK-NEXT:    call void @bar(i128 [[B10]])
+; CHECK-NEXT:    call void @bar(i128 [[B11]])
+; CHECK-NEXT:    call void @bar(i128 [[B12]])
+; CHECK-NEXT:    call void @bar(i128 [[B13]])
+; CHECK-NEXT:    call void @bar(i128 [[B14]])
+; CHECK-NEXT:    call void @bar(i128 [[B15]])
+; CHECK-NEXT:    call void @bar(i128 [[B16]])
+; CHECK-NEXT:    call void @bar(i128 [[B17]])
 ; CHECK-NEXT:    ret void
 ;
-entry:
-  %s2 = shl i128 %s, 1
-  %t1 = add i128 %b1, %s
-  call void @bar(i128 %t1)
-  %t2 = add i128 %b2, %s
-  call void @bar(i128 %t2)
-  %t3 = add i128 %b3, %s
-  call void @bar(i128 %t3)
-  %t4 = add i128 %b4, %s
-  call void @bar(i128 %t4)
-  %u1 = add i128 %b1, %s2
-  call void @bar(i128 %u1)
-  %u2 = add i128 %b2, %s2
-  call void @bar(i128 %u2)
-  %u3 = add i128 %b3, %s2
-  call void @bar(i128 %u3)
-  %u4 = add i128 %b4, %s2
-  call void @bar(i128 %u4)
-  call void @bar(i128 %b1)
-  call void @bar(i128 %b2)
-  call void @bar(i128 %b3)
-  call void @bar(i128 %b4)
-  ret void
-}
-
-; i128 values weigh 4 registers each, so these three bodies peak at 144 with the
-; rewrites applied against 80 without. That straddles the budgets implied by the
-; occupancy attributes, letting the real TTI budget decide with no override.
-
-; No attributes: flat work group size defaults to 1024, so 4 waves per EU and a
-; 128-register budget. 144 exceeds the 115-register safe budget, so filtered.
-define void @default_occupancy(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) {
-; BUDGET32-LABEL: define void @default_occupancy(
-; BUDGET32-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) {
-; BUDGET32-NEXT:  [[ENTRY:.*:]]
-; BUDGET32-NEXT:    [[S2:%.*]] = shl i128 [[S]], 1
-; BUDGET32-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T1]])
-; BUDGET32-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T2]])
-; BUDGET32-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T3]])
-; BUDGET32-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T4]])
-; BUDGET32-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T5]])
-; BUDGET32-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T6]])
-; BUDGET32-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T7]])
-; BUDGET32-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T8]])
-; BUDGET32-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T9]])
-; BUDGET32-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T10]])
-; BUDGET32-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T11]])
-; BUDGET32-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T12]])
-; BUDGET32-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T13]])
-; BUDGET32-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T14]])
-; BUDGET32-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T15]])
-; BUDGET32-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T16]])
-; BUDGET32-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T17]])
-; BUDGET32-NEXT:    [[U1:%.*]] = add i128 [[B1]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U1]])
-; BUDGET32-NEXT:    [[U2:%.*]] = add i128 [[B2]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U2]])
-; BUDGET32-NEXT:    [[U3:%.*]] = add i128 [[B3]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U3]])
-; BUDGET32-NEXT:    [[U4:%.*]] = add i128 [[B4]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U4]])
-; BUDGET32-NEXT:    [[U5:%.*]] = add i128 [[B5]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U5]])
-; BUDGET32-NEXT:    [[U6:%.*]] = add i128 [[B6]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U6]])
-; BUDGET32-NEXT:    [[U7:%.*]] = add i128 [[B7]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U7]])
-; BUDGET32-NEXT:    [[U8:%.*]] = add i128 [[B8]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U8]])
-; BUDGET32-NEXT:    [[U9:%.*]] = add i128 [[B9]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U9]])
-; BUDGET32-NEXT:    [[U10:%.*]] = add i128 [[B10]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U10]])
-; BUDGET32-NEXT:    [[U11:%.*]] = add i128 [[B11]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U11]])
-; BUDGET32-NEXT:    [[U12:%.*]] = add i128 [[B12]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U12]])
-; BUDGET32-NEXT:    [[U13:%.*]] = add i128 [[B13]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U13]])
-; BUDGET32-NEXT:    [[U14:%.*]] = add i128 [[B14]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U14]])
-; BUDGET32-NEXT:    [[U15:%.*]] = add i128 [[B15]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U15]])
-; BUDGET32-NEXT:    [[U16:%.*]] = add i128 [[B16]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U16]])
-; BUDGET32-NEXT:    [[U17:%.*]] = add i128 [[B17]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U17]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B1]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B2]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B3]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B4]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B5]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B6]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B7]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B8]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B9]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B10]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B11]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B12]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B13]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B14]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B15]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B16]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B17]])
-; BUDGET32-NEXT:    ret void
-;
-; ATTRS-LABEL: define void @default_occupancy(
-; ATTRS-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) {
-; ATTRS-NEXT:  [[ENTRY:.*:]]
-; ATTRS-NEXT:    [[S2:%.*]] = shl i128 [[S]], 1
-; ATTRS-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T1]])
-; ATTRS-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T2]])
-; ATTRS-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T3]])
-; ATTRS-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T4]])
-; ATTRS-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T5]])
-; ATTRS-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T6]])
-; ATTRS-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T7]])
-; ATTRS-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T8]])
-; ATTRS-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T9]])
-; ATTRS-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T10]])
-; ATTRS-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T11]])
-; ATTRS-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T12]])
-; ATTRS-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T13]])
-; ATTRS-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T14]])
-; ATTRS-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T15]])
-; ATTRS-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T16]])
-; ATTRS-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T17]])
-; ATTRS-NEXT:    [[U1:%.*]] = add i128 [[B1]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U1]])
-; ATTRS-NEXT:    [[U2:%.*]] = add i128 [[B2]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U2]])
-; ATTRS-NEXT:    [[U3:%.*]] = add i128 [[B3]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U3]])
-; ATTRS-NEXT:    [[U4:%.*]] = add i128 [[B4]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U4]])
-; ATTRS-NEXT:    [[U5:%.*]] = add i128 [[B5]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U5]])
-; ATTRS-NEXT:    [[U6:%.*]] = add i128 [[B6]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U6]])
-; ATTRS-NEXT:    [[U7:%.*]] = add i128 [[B7]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U7]])
-; ATTRS-NEXT:    [[U8:%.*]] = add i128 [[B8]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U8]])
-; ATTRS-NEXT:    [[U9:%.*]] = add i128 [[B9]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U9]])
-; ATTRS-NEXT:    [[U10:%.*]] = add i128 [[B10]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U10]])
-; ATTRS-NEXT:    [[U11:%.*]] = add i128 [[B11]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U11]])
-; ATTRS-NEXT:    [[U12:%.*]] = add i128 [[B12]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U12]])
-; ATTRS-NEXT:    [[U13:%.*]] = add i128 [[B13]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U13]])
-; ATTRS-NEXT:    [[U14:%.*]] = add i128 [[B14]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U14]])
-; ATTRS-NEXT:    [[U15:%.*]] = add i128 [[B15]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U15]])
-; ATTRS-NEXT:    [[U16:%.*]] = add i128 [[B16]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U16]])
-; ATTRS-NEXT:    [[U17:%.*]] = add i128 [[B17]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U17]])
-; ATTRS-NEXT:    call void @bar(i128 [[B1]])
-; ATTRS-NEXT:    call void @bar(i128 [[B2]])
-; ATTRS-NEXT:    call void @bar(i128 [[B3]])
-; ATTRS-NEXT:    call void @bar(i128 [[B4]])
-; ATTRS-NEXT:    call void @bar(i128 [[B5]])
-; ATTRS-NEXT:    call void @bar(i128 [[B6]])
-; ATTRS-NEXT:    call void @bar(i128 [[B7]])
-; ATTRS-NEXT:    call void @bar(i128 [[B8]])
-; ATTRS-NEXT:    call void @bar(i128 [[B9]])
-; ATTRS-NEXT:    call void @bar(i128 [[B10]])
-; ATTRS-NEXT:    call void @bar(i128 [[B11]])
-; ATTRS-NEXT:    call void @bar(i128 [[B12]])
-; ATTRS-NEXT:    call void @bar(i128 [[B13]])
-; ATTRS-NEXT:    call void @bar(i128 [[B14]])
-; ATTRS-NEXT:    call void @bar(i128 [[B15]])
-; ATTRS-NEXT:    call void @bar(i128 [[B16]])
-; ATTRS-NEXT:    call void @bar(i128 [[B17]])
-; ATTRS-NEXT:    ret void
-;
-; NOFILTER-LABEL: define void @default_occupancy(
-; NOFILTER-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) {
-; NOFILTER-NEXT:  [[ENTRY:.*:]]
-; NOFILTER-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T1]])
-; NOFILTER-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T2]])
-; NOFILTER-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T3]])
-; NOFILTER-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T4]])
-; NOFILTER-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T5]])
-; NOFILTER-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T6]])
-; NOFILTER-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T7]])
-; NOFILTER-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T8]])
-; NOFILTER-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T9]])
-; NOFILTER-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T10]])
-; NOFILTER-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T11]])
-; NOFILTER-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T12]])
-; NOFILTER-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T13]])
-; NOFILTER-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T14]])
-; NOFILTER-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T15]])
-; NOFILTER-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T16]])
-; NOFILTER-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T17]])
-; NOFILTER-NEXT:    [[U1:%.*]] = add i128 [[T1]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U1]])
-; NOFILTER-NEXT:    [[U2:%.*]] = add i128 [[T2]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U2]])
-; NOFILTER-NEXT:    [[U3:%.*]] = add i128 [[T3]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U3]])
-; NOFILTER-NEXT:    [[U4:%.*]] = add i128 [[T4]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U4]])
-; NOFILTER-NEXT:    [[U5:%.*]] = add i128 [[T5]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U5]])
-; NOFILTER-NEXT:    [[U6:%.*]] = add i128 [[T6]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U6]])
-; NOFILTER-NEXT:    [[U7:%.*]] = add i128 [[T7]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U7]])
-; NOFILTER-NEXT:    [[U8:%.*]] = add i128 [[T8]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U8]])
-; NOFILTER-NEXT:    [[U9:%.*]] = add i128 [[T9]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U9]])
-; NOFILTER-NEXT:    [[U10:%.*]] = add i128 [[T10]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U10]])
-; NOFILTER-NEXT:    [[U11:%.*]] = add i128 [[T11]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U11]])
-; NOFILTER-NEXT:    [[U12:%.*]] = add i128 [[T12]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U12]])
-; NOFILTER-NEXT:    [[U13:%.*]] = add i128 [[T13]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U13]])
-; NOFILTER-NEXT:    [[U14:%.*]] = add i128 [[T14]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U14]])
-; NOFILTER-NEXT:    [[U15:%.*]] = add i128 [[T15]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U15]])
-; NOFILTER-NEXT:    [[U16:%.*]] = add i128 [[T16]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U16]])
-; NOFILTER-NEXT:    [[U17:%.*]] = add i128 [[T17]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U17]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B1]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B2]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B3]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B4]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B5]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B6]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B7]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B8]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B9]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B10]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B11]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B12]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B13]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B14]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B15]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B16]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B17]])
-; NOFILTER-NEXT:    ret void
-;
 entry:
   %s2 = shl i128 %s, 1
   %t1 = add i128 %b1, %s
@@ -1188,645 +210,96 @@ entry:
   ret void
 }
 
-; A 256-thread work group is 4 wave64s over 4 EUs, so 1 wave per EU and the full
-; 512-register budget. The same 144 now fits, so the rewrites survive.
-define void @small_work_group(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) #0 {
-; BUDGET32-LABEL: define void @small_work_group(
-; BUDGET32-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR0:[0-9]+]] {
-; BUDGET32-NEXT:  [[ENTRY:.*:]]
-; BUDGET32-NEXT:    [[S2:%.*]] = shl i128 [[S]], 1
-; BUDGET32-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T1]])
-; BUDGET32-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T2]])
-; BUDGET32-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T3]])
-; BUDGET32-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T4]])
-; BUDGET32-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T5]])
-; BUDGET32-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T6]])
-; BUDGET32-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T7]])
-; BUDGET32-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T8]])
-; BUDGET32-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T9]])
-; BUDGET32-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T10]])
-; BUDGET32-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T11]])
-; BUDGET32-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T12]])
-; BUDGET32-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T13]])
-; BUDGET32-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T14]])
-; BUDGET32-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T15]])
-; BUDGET32-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T16]])
-; BUDGET32-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T17]])
-; BUDGET32-NEXT:    [[U1:%.*]] = add i128 [[B1]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U1]])
-; BUDGET32-NEXT:    [[U2:%.*]] = add i128 [[B2]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U2]])
-; BUDGET32-NEXT:    [[U3:%.*]] = add i128 [[B3]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U3]])
-; BUDGET32-NEXT:    [[U4:%.*]] = add i128 [[B4]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U4]])
-; BUDGET32-NEXT:    [[U5:%.*]] = add i128 [[B5]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U5]])
-; BUDGET32-NEXT:    [[U6:%.*]] = add i128 [[B6]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U6]])
-; BUDGET32-NEXT:    [[U7:%.*]] = add i128 [[B7]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U7]])
-; BUDGET32-NEXT:    [[U8:%.*]] = add i128 [[B8]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U8]])
-; BUDGET32-NEXT:    [[U9:%.*]] = add i128 [[B9]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U9]])
-; BUDGET32-NEXT:    [[U10:%.*]] = add i128 [[B10]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U10]])
-; BUDGET32-NEXT:    [[U11:%.*]] = add i128 [[B11]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U11]])
-; BUDGET32-NEXT:    [[U12:%.*]] = add i128 [[B12]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U12]])
-; BUDGET32-NEXT:    [[U13:%.*]] = add i128 [[B13]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U13]])
-; BUDGET32-NEXT:    [[U14:%.*]] = add i128 [[B14]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U14]])
-; BUDGET32-NEXT:    [[U15:%.*]] = add i128 [[B15]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U15]])
-; BUDGET32-NEXT:    [[U16:%.*]] = add i128 [[B16]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U16]])
-; BUDGET32-NEXT:    [[U17:%.*]] = add i128 [[B17]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U17]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B1]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B2]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B3]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B4]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B5]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B6]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B7]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B8]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B9]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B10]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B11]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B12]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B13]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B14]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B15]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B16]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B17]])
-; BUDGET32-NEXT:    ret void
-;
-; ATTRS-LABEL: define void @small_work_group(
-; ATTRS-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR0:[0-9]+]] {
-; ATTRS-NEXT:  [[ENTRY:.*:]]
-; ATTRS-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T1]])
-; ATTRS-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T2]])
-; ATTRS-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T3]])
-; ATTRS-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T4]])
-; ATTRS-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T5]])
-; ATTRS-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T6]])
-; ATTRS-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T7]])
-; ATTRS-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T8]])
-; ATTRS-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T9]])
-; ATTRS-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T10]])
-; ATTRS-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T11]])
-; ATTRS-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T12]])
-; ATTRS-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T13]])
-; ATTRS-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T14]])
-; ATTRS-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T15]])
-; ATTRS-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T16]])
-; ATTRS-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T17]])
-; ATTRS-NEXT:    [[U1:%.*]] = add i128 [[T1]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U1]])
-; ATTRS-NEXT:    [[U2:%.*]] = add i128 [[T2]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U2]])
-; ATTRS-NEXT:    [[U3:%.*]] = add i128 [[T3]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U3]])
-; ATTRS-NEXT:    [[U4:%.*]] = add i128 [[T4]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U4]])
-; ATTRS-NEXT:    [[U5:%.*]] = add i128 [[T5]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U5]])
-; ATTRS-NEXT:    [[U6:%.*]] = add i128 [[T6]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U6]])
-; ATTRS-NEXT:    [[U7:%.*]] = add i128 [[T7]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U7]])
-; ATTRS-NEXT:    [[U8:%.*]] = add i128 [[T8]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U8]])
-; ATTRS-NEXT:    [[U9:%.*]] = add i128 [[T9]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U9]])
-; ATTRS-NEXT:    [[U10:%.*]] = add i128 [[T10]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U10]])
-; ATTRS-NEXT:    [[U11:%.*]] = add i128 [[T11]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U11]])
-; ATTRS-NEXT:    [[U12:%.*]] = add i128 [[T12]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U12]])
-; ATTRS-NEXT:    [[U13:%.*]] = add i128 [[T13]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U13]])
-; ATTRS-NEXT:    [[U14:%.*]] = add i128 [[T14]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U14]])
-; ATTRS-NEXT:    [[U15:%.*]] = add i128 [[T15]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U15]])
-; ATTRS-NEXT:    [[U16:%.*]] = add i128 [[T16]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U16]])
-; ATTRS-NEXT:    [[U17:%.*]] = add i128 [[T17]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[U17]])
-; ATTRS-NEXT:    call void @bar(i128 [[B1]])
-; ATTRS-NEXT:    call void @bar(i128 [[B2]])
-; ATTRS-NEXT:    call void @bar(i128 [[B3]])
-; ATTRS-NEXT:    call void @bar(i128 [[B4]])
-; ATTRS-NEXT:    call void @bar(i128 [[B5]])
-; ATTRS-NEXT:    call void @bar(i128 [[B6]])
-; ATTRS-NEXT:    call void @bar(i128 [[B7]])
-; ATTRS-NEXT:    call void @bar(i128 [[B8]])
-; ATTRS-NEXT:    call void @bar(i128 [[B9]])
-; ATTRS-NEXT:    call void @bar(i128 [[B10]])
-; ATTRS-NEXT:    call void @bar(i128 [[B11]])
-; ATTRS-NEXT:    call void @bar(i128 [[B12]])
-; ATTRS-NEXT:    call void @bar(i128 [[B13]])
-; ATTRS-NEXT:    call void @bar(i128 [[B14]])
-; ATTRS-NEXT:    call void @bar(i128 [[B15]])
-; ATTRS-NEXT:    call void @bar(i128 [[B16]])
-; ATTRS-NEXT:    call void @bar(i128 [[B17]])
-; ATTRS-NEXT:    ret void
-;
-; NOFILTER-LABEL: define void @small_work_group(
-; NOFILTER-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR0:[0-9]+]] {
-; NOFILTER-NEXT:  [[ENTRY:.*:]]
-; NOFILTER-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T1]])
-; NOFILTER-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T2]])
-; NOFILTER-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T3]])
-; NOFILTER-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T4]])
-; NOFILTER-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T5]])
-; NOFILTER-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T6]])
-; NOFILTER-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T7]])
-; NOFILTER-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T8]])
-; NOFILTER-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T9]])
-; NOFILTER-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T10]])
-; NOFILTER-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T11]])
-; NOFILTER-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T12]])
-; NOFILTER-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T13]])
-; NOFILTER-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T14]])
-; NOFILTER-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T15]])
-; NOFILTER-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T16]])
-; NOFILTER-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T17]])
-; NOFILTER-NEXT:    [[U1:%.*]] = add i128 [[T1]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U1]])
-; NOFILTER-NEXT:    [[U2:%.*]] = add i128 [[T2]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U2]])
-; NOFILTER-NEXT:    [[U3:%.*]] = add i128 [[T3]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U3]])
-; NOFILTER-NEXT:    [[U4:%.*]] = add i128 [[T4]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U4]])
-; NOFILTER-NEXT:    [[U5:%.*]] = add i128 [[T5]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U5]])
-; NOFILTER-NEXT:    [[U6:%.*]] = add i128 [[T6]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U6]])
-; NOFILTER-NEXT:    [[U7:%.*]] = add i128 [[T7]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U7]])
-; NOFILTER-NEXT:    [[U8:%.*]] = add i128 [[T8]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U8]])
-; NOFILTER-NEXT:    [[U9:%.*]] = add i128 [[T9]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U9]])
-; NOFILTER-NEXT:    [[U10:%.*]] = add i128 [[T10]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U10]])
-; NOFILTER-NEXT:    [[U11:%.*]] = add i128 [[T11]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U11]])
-; NOFILTER-NEXT:    [[U12:%.*]] = add i128 [[T12]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U12]])
-; NOFILTER-NEXT:    [[U13:%.*]] = add i128 [[T13]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U13]])
-; NOFILTER-NEXT:    [[U14:%.*]] = add i128 [[T14]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U14]])
-; NOFILTER-NEXT:    [[U15:%.*]] = add i128 [[T15]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U15]])
-; NOFILTER-NEXT:    [[U16:%.*]] = add i128 [[T16]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U16]])
-; NOFILTER-NEXT:    [[U17:%.*]] = add i128 [[T17]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U17]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B1]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B2]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B3]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B4]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B5]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B6]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B7]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B8]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B9]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B10]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B11]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B12]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B13]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B14]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B15]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B16]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B17]])
-; NOFILTER-NEXT:    ret void
-;
-entry:
-  %s2 = shl i128 %s, 1
-  %t1 = add i128 %b1, %s
-  call void @bar(i128 %t1)
-  %t2 = add i128 %b2, %s
-  call void @bar(i128 %t2)
-  %t3 = add i128 %b3, %s
-  call void @bar(i128 %t3)
-  %t4 = add i128 %b4, %s
-  call void @bar(i128 %t4)
-  %t5 = add i128 %b5, %s
-  call void @bar(i128 %t5)
-  %t6 = add i128 %b6, %s
-  call void @bar(i128 %t6)
-  %t7 = add i128 %b7, %s
-  call void @bar(i128 %t7)
-  %t8 = add i128 %b8, %s
-  call void @bar(i128 %t8)
-  %t9 = add i128 %b9, %s
-  call void @bar(i128 %t9)
-  %t10 = add i128 %b10, %s
-  call void @bar(i128 %t10)
-  %t11 = add i128 %b11, %s
-  call void @bar(i128 %t11)
-  %t12 = add i128 %b12, %s
-  call void @bar(i128 %t12)
-  %t13 = add i128 %b13, %s
-  call void @bar(i128 %t13)
-  %t14 = add i128 %b14, %s
-  call void @bar(i128 %t14)
-  %t15 = add i128 %b15, %s
-  call void @bar(i128 %t15)
-  %t16 = add i128 %b16, %s
-  call void @bar(i128 %t16)
-  %t17 = add i128 %b17, %s
-  call void @bar(i128 %t17)
-  %u1 = add i128 %b1, %s2
-  call void @bar(i128 %u1)
-  %u2 = add i128 %b2, %s2
-  call void @bar(i128 %u2)
-  %u3 = add i128 %b3, %s2
-  call void @bar(i128 %u3)
-  %u4 = add i128 %b4, %s2
-  call void @bar(i128 %u4)
-  %u5 = add i128 %b5, %s2
-  call void @bar(i128 %u5)
-  %u6 = add i128 %b6, %s2
-  call void @bar(i128 %u6)
-  %u7 = add i128 %b7, %s2
-  call void @bar(i128 %u7)
-  %u8 = add i128 %b8, %s2
-  call void @bar(i128 %u8)
-  %u9 = add i128 %b9, %s2
-  call void @bar(i128 %u9)
-  %u10 = add i128 %b10, %s2
-  call void @bar(i128 %u10)
-  %u11 = add i128 %b11, %s2
-  call void @bar(i128 %u11)
-  %u12 = add i128 %b12, %s2
-  call void @bar(i128 %u12)
-  %u13 = add i128 %b13, %s2
-  call void @bar(i128 %u13)
-  %u14 = add i128 %b14, %s2
-  call void @bar(i128 %u14)
-  %u15 = add i128 %b15, %s2
-  call void @bar(i128 %u15)
-  %u16 = add i128 %b16, %s2
-  call void @bar(i128 %u16)
-  %u17 = add i128 %b17, %s2
-  call void @bar(i128 %u17)
-  call void @bar(i128 %b1)
-  call void @bar(i128 %b2)
-  call void @bar(i128 %b3)
-  call void @bar(i128 %b4)
-  call void @bar(i128 %b5)
-  call void @bar(i128 %b6)
-  call void @bar(i128 %b7)
-  call void @bar(i128 %b8)
-  call void @bar(i128 %b9)
-  call void @bar(i128 %b10)
-  call void @bar(i128 %b11)
-  call void @bar(i128 %b12)
-  call void @bar(i128 %b13)
-  call void @bar(i128 %b14)
-  call void @bar(i128 %b15)
-  call void @bar(i128 %b16)
-  call void @bar(i128 %b17)
-  ret void
-}
-
-; Same small work group, but asking for 8 waves per EU drops the budget to
-; 512/8 = 64, so 144 is over again and the rewrites are dropped. This is the
-; pair that shows waves-per-eu overriding what the work group size allows.
-define void @small_wg_high_occ(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) #1 {
-; BUDGET32-LABEL: define void @small_wg_high_occ(
-; BUDGET32-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR1:[0-9]+]] {
-; BUDGET32-NEXT:  [[ENTRY:.*:]]
-; BUDGET32-NEXT:    [[S2:%.*]] = shl i128 [[S]], 1
-; BUDGET32-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T1]])
-; BUDGET32-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T2]])
-; BUDGET32-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T3]])
-; BUDGET32-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T4]])
-; BUDGET32-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T5]])
-; BUDGET32-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T6]])
-; BUDGET32-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T7]])
-; BUDGET32-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T8]])
-; BUDGET32-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T9]])
-; BUDGET32-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T10]])
-; BUDGET32-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T11]])
-; BUDGET32-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T12]])
-; BUDGET32-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T13]])
-; BUDGET32-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T14]])
-; BUDGET32-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T15]])
-; BUDGET32-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T16]])
-; BUDGET32-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
-; BUDGET32-NEXT:    call void @bar(i128 [[T17]])
-; BUDGET32-NEXT:    [[U1:%.*]] = add i128 [[B1]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U1]])
-; BUDGET32-NEXT:    [[U2:%.*]] = add i128 [[B2]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U2]])
-; BUDGET32-NEXT:    [[U3:%.*]] = add i128 [[B3]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U3]])
-; BUDGET32-NEXT:    [[U4:%.*]] = add i128 [[B4]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U4]])
-; BUDGET32-NEXT:    [[U5:%.*]] = add i128 [[B5]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U5]])
-; BUDGET32-NEXT:    [[U6:%.*]] = add i128 [[B6]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U6]])
-; BUDGET32-NEXT:    [[U7:%.*]] = add i128 [[B7]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U7]])
-; BUDGET32-NEXT:    [[U8:%.*]] = add i128 [[B8]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U8]])
-; BUDGET32-NEXT:    [[U9:%.*]] = add i128 [[B9]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U9]])
-; BUDGET32-NEXT:    [[U10:%.*]] = add i128 [[B10]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U10]])
-; BUDGET32-NEXT:    [[U11:%.*]] = add i128 [[B11]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U11]])
-; BUDGET32-NEXT:    [[U12:%.*]] = add i128 [[B12]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U12]])
-; BUDGET32-NEXT:    [[U13:%.*]] = add i128 [[B13]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U13]])
-; BUDGET32-NEXT:    [[U14:%.*]] = add i128 [[B14]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U14]])
-; BUDGET32-NEXT:    [[U15:%.*]] = add i128 [[B15]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U15]])
-; BUDGET32-NEXT:    [[U16:%.*]] = add i128 [[B16]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U16]])
-; BUDGET32-NEXT:    [[U17:%.*]] = add i128 [[B17]], [[S2]]
-; BUDGET32-NEXT:    call void @bar(i128 [[U17]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B1]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B2]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B3]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B4]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B5]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B6]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B7]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B8]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B9]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B10]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B11]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B12]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B13]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B14]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B15]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B16]])
-; BUDGET32-NEXT:    call void @bar(i128 [[B17]])
-; BUDGET32-NEXT:    ret void
-;
-; ATTRS-LABEL: define void @small_wg_high_occ(
-; ATTRS-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR1:[0-9]+]] {
-; ATTRS-NEXT:  [[ENTRY:.*:]]
-; ATTRS-NEXT:    [[S2:%.*]] = shl i128 [[S]], 1
-; ATTRS-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T1]])
-; ATTRS-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T2]])
-; ATTRS-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T3]])
-; ATTRS-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T4]])
-; ATTRS-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T5]])
-; ATTRS-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T6]])
-; ATTRS-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T7]])
-; ATTRS-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T8]])
-; ATTRS-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T9]])
-; ATTRS-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T10]])
-; ATTRS-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T11]])
-; ATTRS-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T12]])
-; ATTRS-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T13]])
-; ATTRS-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T14]])
-; ATTRS-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T15]])
-; ATTRS-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T16]])
-; ATTRS-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
-; ATTRS-NEXT:    call void @bar(i128 [[T17]])
-; ATTRS-NEXT:    [[U1:%.*]] = add i128 [[B1]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U1]])
-; ATTRS-NEXT:    [[U2:%.*]] = add i128 [[B2]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U2]])
-; ATTRS-NEXT:    [[U3:%.*]] = add i128 [[B3]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U3]])
-; ATTRS-NEXT:    [[U4:%.*]] = add i128 [[B4]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U4]])
-; ATTRS-NEXT:    [[U5:%.*]] = add i128 [[B5]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U5]])
-; ATTRS-NEXT:    [[U6:%.*]] = add i128 [[B6]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U6]])
-; ATTRS-NEXT:    [[U7:%.*]] = add i128 [[B7]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U7]])
-; ATTRS-NEXT:    [[U8:%.*]] = add i128 [[B8]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U8]])
-; ATTRS-NEXT:    [[U9:%.*]] = add i128 [[B9]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U9]])
-; ATTRS-NEXT:    [[U10:%.*]] = add i128 [[B10]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U10]])
-; ATTRS-NEXT:    [[U11:%.*]] = add i128 [[B11]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U11]])
-; ATTRS-NEXT:    [[U12:%.*]] = add i128 [[B12]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U12]])
-; ATTRS-NEXT:    [[U13:%.*]] = add i128 [[B13]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U13]])
-; ATTRS-NEXT:    [[U14:%.*]] = add i128 [[B14]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U14]])
-; ATTRS-NEXT:    [[U15:%.*]] = add i128 [[B15]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U15]])
-; ATTRS-NEXT:    [[U16:%.*]] = add i128 [[B16]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U16]])
-; ATTRS-NEXT:    [[U17:%.*]] = add i128 [[B17]], [[S2]]
-; ATTRS-NEXT:    call void @bar(i128 [[U17]])
-; ATTRS-NEXT:    call void @bar(i128 [[B1]])
-; ATTRS-NEXT:    call void @bar(i128 [[B2]])
-; ATTRS-NEXT:    call void @bar(i128 [[B3]])
-; ATTRS-NEXT:    call void @bar(i128 [[B4]])
-; ATTRS-NEXT:    call void @bar(i128 [[B5]])
-; ATTRS-NEXT:    call void @bar(i128 [[B6]])
-; ATTRS-NEXT:    call void @bar(i128 [[B7]])
-; ATTRS-NEXT:    call void @bar(i128 [[B8]])
-; ATTRS-NEXT:    call void @bar(i128 [[B9]])
-; ATTRS-NEXT:    call void @bar(i128 [[B10]])
-; ATTRS-NEXT:    call void @bar(i128 [[B11]])
-; ATTRS-NEXT:    call void @bar(i128 [[B12]])
-; ATTRS-NEXT:    call void @bar(i128 [[B13]])
-; ATTRS-NEXT:    call void @bar(i128 [[B14]])
-; ATTRS-NEXT:    call void @bar(i128 [[B15]])
-; ATTRS-NEXT:    call void @bar(i128 [[B16]])
-; ATTRS-NEXT:    call void @bar(i128 [[B17]])
-; ATTRS-NEXT:    ret void
-;
-; NOFILTER-LABEL: define void @small_wg_high_occ(
-; NOFILTER-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR1:[0-9]+]] {
-; NOFILTER-NEXT:  [[ENTRY:.*:]]
-; NOFILTER-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T1]])
-; NOFILTER-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T2]])
-; NOFILTER-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T3]])
-; NOFILTER-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T4]])
-; NOFILTER-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T5]])
-; NOFILTER-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T6]])
-; NOFILTER-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T7]])
-; NOFILTER-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T8]])
-; NOFILTER-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T9]])
-; NOFILTER-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T10]])
-; NOFILTER-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T11]])
-; NOFILTER-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T12]])
-; NOFILTER-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T13]])
-; NOFILTER-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T14]])
-; NOFILTER-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T15]])
-; NOFILTER-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T16]])
-; NOFILTER-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[T17]])
-; NOFILTER-NEXT:    [[U1:%.*]] = add i128 [[T1]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U1]])
-; NOFILTER-NEXT:    [[U2:%.*]] = add i128 [[T2]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U2]])
-; NOFILTER-NEXT:    [[U3:%.*]] = add i128 [[T3]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U3]])
-; NOFILTER-NEXT:    [[U4:%.*]] = add i128 [[T4]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U4]])
-; NOFILTER-NEXT:    [[U5:%.*]] = add i128 [[T5]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U5]])
-; NOFILTER-NEXT:    [[U6:%.*]] = add i128 [[T6]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U6]])
-; NOFILTER-NEXT:    [[U7:%.*]] = add i128 [[T7]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U7]])
-; NOFILTER-NEXT:    [[U8:%.*]] = add i128 [[T8]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U8]])
-; NOFILTER-NEXT:    [[U9:%.*]] = add i128 [[T9]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U9]])
-; NOFILTER-NEXT:    [[U10:%.*]] = add i128 [[T10]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U10]])
-; NOFILTER-NEXT:    [[U11:%.*]] = add i128 [[T11]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U11]])
-; NOFILTER-NEXT:    [[U12:%.*]] = add i128 [[T12]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U12]])
-; NOFILTER-NEXT:    [[U13:%.*]] = add i128 [[T13]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U13]])
-; NOFILTER-NEXT:    [[U14:%.*]] = add i128 [[T14]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U14]])
-; NOFILTER-NEXT:    [[U15:%.*]] = add i128 [[T15]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U15]])
-; NOFILTER-NEXT:    [[U16:%.*]] = add i128 [[T16]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U16]])
-; NOFILTER-NEXT:    [[U17:%.*]] = add i128 [[T17]], [[S]]
-; NOFILTER-NEXT:    call void @bar(i128 [[U17]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B1]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B2]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B3]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B4]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B5]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B6]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B7]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B8]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B9]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B10]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B11]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B12]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B13]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B14]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B15]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B16]])
-; NOFILTER-NEXT:    call void @bar(i128 [[B17]])
-; NOFILTER-NEXT:    ret void
+define void @peak_below_budget(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) #0 {
+; CHECK-LABEL: define void @peak_below_budget(
+; CHECK-SAME: i128 [[S:%.*]], i128 [[B1:%.*]], i128 [[B2:%.*]], i128 [[B3:%.*]], i128 [[B4:%.*]], i128 [[B5:%.*]], i128 [[B6:%.*]], i128 [[B7:%.*]], i128 [[B8:%.*]], i128 [[B9:%.*]], i128 [[B10:%.*]], i128 [[B11:%.*]], i128 [[B12:%.*]], i128 [[B13:%.*]], i128 [[B14:%.*]], i128 [[B15:%.*]], i128 [[B16:%.*]], i128 [[B17:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[T1:%.*]] = add i128 [[B1]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T1]])
+; CHECK-NEXT:    [[T2:%.*]] = add i128 [[B2]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T2]])
+; CHECK-NEXT:    [[T3:%.*]] = add i128 [[B3]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T3]])
+; CHECK-NEXT:    [[T4:%.*]] = add i128 [[B4]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T4]])
+; CHECK-NEXT:    [[T5:%.*]] = add i128 [[B5]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T5]])
+; CHECK-NEXT:    [[T6:%.*]] = add i128 [[B6]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T6]])
+; CHECK-NEXT:    [[T7:%.*]] = add i128 [[B7]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T7]])
+; CHECK-NEXT:    [[T8:%.*]] = add i128 [[B8]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T8]])
+; CHECK-NEXT:    [[T9:%.*]] = add i128 [[B9]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T9]])
+; CHECK-NEXT:    [[T10:%.*]] = add i128 [[B10]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T10]])
+; CHECK-NEXT:    [[T11:%.*]] = add i128 [[B11]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T11]])
+; CHECK-NEXT:    [[T12:%.*]] = add i128 [[B12]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T12]])
+; CHECK-NEXT:    [[T13:%.*]] = add i128 [[B13]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T13]])
+; CHECK-NEXT:    [[T14:%.*]] = add i128 [[B14]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T14]])
+; CHECK-NEXT:    [[T15:%.*]] = add i128 [[B15]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T15]])
+; CHECK-NEXT:    [[T16:%.*]] = add i128 [[B16]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T16]])
+; CHECK-NEXT:    [[T17:%.*]] = add i128 [[B17]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[T17]])
+; CHECK-NEXT:    [[U1:%.*]] = add i128 [[T1]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U1]])
+; CHECK-NEXT:    [[U2:%.*]] = add i128 [[T2]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U2]])
+; CHECK-NEXT:    [[U3:%.*]] = add i128 [[T3]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U3]])
+; CHECK-NEXT:    [[U4:%.*]] = add i128 [[T4]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U4]])
+; CHECK-NEXT:    [[U5:%.*]] = add i128 [[T5]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U5]])
+; CHECK-NEXT:    [[U6:%.*]] = add i128 [[T6]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U6]])
+; CHECK-NEXT:    [[U7:%.*]] = add i128 [[T7]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U7]])
+; CHECK-NEXT:    [[U8:%.*]] = add i128 [[T8]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U8]])
+; CHECK-NEXT:    [[U9:%.*]] = add i128 [[T9]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U9]])
+; CHECK-NEXT:    [[U10:%.*]] = add i128 [[T10]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U10]])
+; CHECK-NEXT:    [[U11:%.*]] = add i128 [[T11]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U11]])
+; CHECK-NEXT:    [[U12:%.*]] = add i128 [[T12]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U12]])
+; CHECK-NEXT:    [[U13:%.*]] = add i128 [[T13]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U13]])
+; CHECK-NEXT:    [[U14:%.*]] = add i128 [[T14]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U14]])
+; CHECK-NEXT:    [[U15:%.*]] = add i128 [[T15]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U15]])
+; CHECK-NEXT:    [[U16:%.*]] = add i128 [[T16]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U16]])
+; CHECK-NEXT:    [[U17:%.*]] = add i128 [[T17]], [[S]]
+; CHECK-NEXT:    call void @bar(i128 [[U17]])
+; CHECK-NEXT:    call void @bar(i128 [[B1]])
+; CHECK-NEXT:    call void @bar(i128 [[B2]])
+; CHECK-NEXT:    call void @bar(i128 [[B3]])
+; CHECK-NEXT:    call void @bar(i128 [[B4]])
+; CHECK-NEXT:    call void @bar(i128 [[B5]])
+; CHECK-NEXT:    call void @bar(i128 [[B6]])
+; CHECK-NEXT:    call void @bar(i128 [[B7]])
+; CHECK-NEXT:    call void @bar(i128 [[B8]])
+; CHECK-NEXT:    call void @bar(i128 [[B9]])
+; CHECK-NEXT:    call void @bar(i128 [[B10]])
+; CHECK-NEXT:    call void @bar(i128 [[B11]])
+; CHECK-NEXT:    call void @bar(i128 [[B12]])
+; CHECK-NEXT:    call void @bar(i128 [[B13]])
+; CHECK-NEXT:    call void @bar(i128 [[B14]])
+; CHECK-NEXT:    call void @bar(i128 [[B15]])
+; CHECK-NEXT:    call void @bar(i128 [[B16]])
+; CHECK-NEXT:    call void @bar(i128 [[B17]])
+; CHECK-NEXT:    ret void
 ;
 entry:
   %s2 = shl i128 %s, 1
@@ -1918,5 +391,4 @@ entry:
   ret void
 }
 
-attributes #0 = { "amdgpu-flat-work-group-size"="1,256" }
-attributes #1 = { "amdgpu-flat-work-group-size"="1,256" "amdgpu-waves-per-eu"="8" }
+attributes #0 = { "amdgpu-flat-work-group-size"="1,256" }
\ No newline at end of file
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
index c534ecb8a3fda..32761d2922acd 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
@@ -1,296 +1,114 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt < %s -passes=slsr -S | FileCheck %s --check-prefixes=CHECK,NOBUDGET
-; RUN: opt < %s -passes=slsr -S -slsr-reg-budget=32 | FileCheck %s --check-prefixes=CHECK,BUDGET
+; REQUIRES: asserts
+; RUN: opt -passes=slsr -stats -disable-output <%s 2>&1 | FileCheck %s
+
+; CHECK-NOT: Number of blocks whose rewrites SLSR skipped due to register pressure
 
 ; The register-pressure filter needs a register budget to compare against, and
 ; the generic TargetTransformInfo has none: getRegisterBudget() returns
 ; std::nullopt unless a target implements it. There is no target triple here, so
-; the filter returns early without even running its liveness or pressure
-; analyses, and every rewrite stands no matter how much pressure it adds.
+; RPFilter::run() returns early, before it even computes liveness or pressure,
+; and every rewrite stands.
 ;
-; @many_bases_overlapping is the same shape used in AMDGPU/slsr-rp-filter.ll:
-; 17 distinct bases whose rewrites take the block's peak pressure from 20 to 36
-; registers. Passing -slsr-reg-budget=32 supplies the missing budget, and with
-; the 0.9 safe fraction leaving 28 registers the rewrites are then dropped.
+; @many_basises_overlapping is the same body used by
+; @peak_above_budget in AMDGPU/slsr-rp-filter.ll: 17 distinct basises whose
+; rewrites take the block's peak pressure from 80 registers to 144. Under an
+; AMDGPU triple that exceeds the budget and the rewrites are dropped.
 ;
-; So the two runs differ only in whether a budget exists, which is what pins the
-; early return down: without one the filter is inert on targets that cannot
-; report a register count.
+; The check is on the statistic rather than on the IR because LLVM only prints a
+; statistic that was incremented at least once, so a missing counter means the
+; filter never skipped a block.
 
 target datalayout = "e-i64:64-v16:16-v32:32-n16:32:64"
 
-declare void @foo(i32)
+declare void @bar(i128)
 
-define void @many_bases_overlapping(i32 %s, i32 %b1, i32 %b2, i32 %b3, i32 %b4, i32 %b5, i32 %b6, i32 %b7, i32 %b8, i32 %b9, i32 %b10, i32 %b11, i32 %b12, i32 %b13, i32 %b14, i32 %b15, i32 %b16, i32 %b17) {
-; NOBUDGET-LABEL: define void @many_bases_overlapping(
-; NOBUDGET-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; NOBUDGET-NEXT:  [[ENTRY:.*:]]
-; NOBUDGET-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T1]])
-; NOBUDGET-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T2]])
-; NOBUDGET-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T3]])
-; NOBUDGET-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T4]])
-; NOBUDGET-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T5]])
-; NOBUDGET-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T6]])
-; NOBUDGET-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T7]])
-; NOBUDGET-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T8]])
-; NOBUDGET-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T9]])
-; NOBUDGET-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T10]])
-; NOBUDGET-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T11]])
-; NOBUDGET-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T12]])
-; NOBUDGET-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T13]])
-; NOBUDGET-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T14]])
-; NOBUDGET-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T15]])
-; NOBUDGET-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T16]])
-; NOBUDGET-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[T17]])
-; NOBUDGET-NEXT:    [[U1:%.*]] = add i32 [[T1]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U1]])
-; NOBUDGET-NEXT:    [[U2:%.*]] = add i32 [[T2]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U2]])
-; NOBUDGET-NEXT:    [[U3:%.*]] = add i32 [[T3]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U3]])
-; NOBUDGET-NEXT:    [[U4:%.*]] = add i32 [[T4]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U4]])
-; NOBUDGET-NEXT:    [[U5:%.*]] = add i32 [[T5]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U5]])
-; NOBUDGET-NEXT:    [[U6:%.*]] = add i32 [[T6]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U6]])
-; NOBUDGET-NEXT:    [[U7:%.*]] = add i32 [[T7]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U7]])
-; NOBUDGET-NEXT:    [[U8:%.*]] = add i32 [[T8]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U8]])
-; NOBUDGET-NEXT:    [[U9:%.*]] = add i32 [[T9]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U9]])
-; NOBUDGET-NEXT:    [[U10:%.*]] = add i32 [[T10]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U10]])
-; NOBUDGET-NEXT:    [[U11:%.*]] = add i32 [[T11]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U11]])
-; NOBUDGET-NEXT:    [[U12:%.*]] = add i32 [[T12]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U12]])
-; NOBUDGET-NEXT:    [[U13:%.*]] = add i32 [[T13]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U13]])
-; NOBUDGET-NEXT:    [[U14:%.*]] = add i32 [[T14]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U14]])
-; NOBUDGET-NEXT:    [[U15:%.*]] = add i32 [[T15]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U15]])
-; NOBUDGET-NEXT:    [[U16:%.*]] = add i32 [[T16]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U16]])
-; NOBUDGET-NEXT:    [[U17:%.*]] = add i32 [[T17]], [[S]]
-; NOBUDGET-NEXT:    call void @foo(i32 [[U17]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B1]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B2]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B3]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B4]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B5]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B6]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B7]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B8]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B9]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B10]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B11]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B12]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B13]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B14]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B15]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B16]])
-; NOBUDGET-NEXT:    call void @foo(i32 [[B17]])
-; NOBUDGET-NEXT:    ret void
-;
-; BUDGET-LABEL: define void @many_bases_overlapping(
-; BUDGET-SAME: i32 [[S:%.*]], i32 [[B1:%.*]], i32 [[B2:%.*]], i32 [[B3:%.*]], i32 [[B4:%.*]], i32 [[B5:%.*]], i32 [[B6:%.*]], i32 [[B7:%.*]], i32 [[B8:%.*]], i32 [[B9:%.*]], i32 [[B10:%.*]], i32 [[B11:%.*]], i32 [[B12:%.*]], i32 [[B13:%.*]], i32 [[B14:%.*]], i32 [[B15:%.*]], i32 [[B16:%.*]], i32 [[B17:%.*]]) {
-; BUDGET-NEXT:  [[ENTRY:.*:]]
-; BUDGET-NEXT:    [[S2:%.*]] = shl i32 [[S]], 1
-; BUDGET-NEXT:    [[T1:%.*]] = add i32 [[B1]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T1]])
-; BUDGET-NEXT:    [[T2:%.*]] = add i32 [[B2]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T2]])
-; BUDGET-NEXT:    [[T3:%.*]] = add i32 [[B3]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T3]])
-; BUDGET-NEXT:    [[T4:%.*]] = add i32 [[B4]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T4]])
-; BUDGET-NEXT:    [[T5:%.*]] = add i32 [[B5]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T5]])
-; BUDGET-NEXT:    [[T6:%.*]] = add i32 [[B6]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T6]])
-; BUDGET-NEXT:    [[T7:%.*]] = add i32 [[B7]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T7]])
-; BUDGET-NEXT:    [[T8:%.*]] = add i32 [[B8]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T8]])
-; BUDGET-NEXT:    [[T9:%.*]] = add i32 [[B9]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T9]])
-; BUDGET-NEXT:    [[T10:%.*]] = add i32 [[B10]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T10]])
-; BUDGET-NEXT:    [[T11:%.*]] = add i32 [[B11]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T11]])
-; BUDGET-NEXT:    [[T12:%.*]] = add i32 [[B12]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T12]])
-; BUDGET-NEXT:    [[T13:%.*]] = add i32 [[B13]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T13]])
-; BUDGET-NEXT:    [[T14:%.*]] = add i32 [[B14]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T14]])
-; BUDGET-NEXT:    [[T15:%.*]] = add i32 [[B15]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T15]])
-; BUDGET-NEXT:    [[T16:%.*]] = add i32 [[B16]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T16]])
-; BUDGET-NEXT:    [[T17:%.*]] = add i32 [[B17]], [[S]]
-; BUDGET-NEXT:    call void @foo(i32 [[T17]])
-; BUDGET-NEXT:    [[U1:%.*]] = add i32 [[B1]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U1]])
-; BUDGET-NEXT:    [[U2:%.*]] = add i32 [[B2]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U2]])
-; BUDGET-NEXT:    [[U3:%.*]] = add i32 [[B3]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U3]])
-; BUDGET-NEXT:    [[U4:%.*]] = add i32 [[B4]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U4]])
-; BUDGET-NEXT:    [[U5:%.*]] = add i32 [[B5]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U5]])
-; BUDGET-NEXT:    [[U6:%.*]] = add i32 [[B6]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U6]])
-; BUDGET-NEXT:    [[U7:%.*]] = add i32 [[B7]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U7]])
-; BUDGET-NEXT:    [[U8:%.*]] = add i32 [[B8]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U8]])
-; BUDGET-NEXT:    [[U9:%.*]] = add i32 [[B9]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U9]])
-; BUDGET-NEXT:    [[U10:%.*]] = add i32 [[B10]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U10]])
-; BUDGET-NEXT:    [[U11:%.*]] = add i32 [[B11]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U11]])
-; BUDGET-NEXT:    [[U12:%.*]] = add i32 [[B12]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U12]])
-; BUDGET-NEXT:    [[U13:%.*]] = add i32 [[B13]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U13]])
-; BUDGET-NEXT:    [[U14:%.*]] = add i32 [[B14]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U14]])
-; BUDGET-NEXT:    [[U15:%.*]] = add i32 [[B15]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U15]])
-; BUDGET-NEXT:    [[U16:%.*]] = add i32 [[B16]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U16]])
-; BUDGET-NEXT:    [[U17:%.*]] = add i32 [[B17]], [[S2]]
-; BUDGET-NEXT:    call void @foo(i32 [[U17]])
-; BUDGET-NEXT:    call void @foo(i32 [[B1]])
-; BUDGET-NEXT:    call void @foo(i32 [[B2]])
-; BUDGET-NEXT:    call void @foo(i32 [[B3]])
-; BUDGET-NEXT:    call void @foo(i32 [[B4]])
-; BUDGET-NEXT:    call void @foo(i32 [[B5]])
-; BUDGET-NEXT:    call void @foo(i32 [[B6]])
-; BUDGET-NEXT:    call void @foo(i32 [[B7]])
-; BUDGET-NEXT:    call void @foo(i32 [[B8]])
-; BUDGET-NEXT:    call void @foo(i32 [[B9]])
-; BUDGET-NEXT:    call void @foo(i32 [[B10]])
-; BUDGET-NEXT:    call void @foo(i32 [[B11]])
-; BUDGET-NEXT:    call void @foo(i32 [[B12]])
-; BUDGET-NEXT:    call void @foo(i32 [[B13]])
-; BUDGET-NEXT:    call void @foo(i32 [[B14]])
-; BUDGET-NEXT:    call void @foo(i32 [[B15]])
-; BUDGET-NEXT:    call void @foo(i32 [[B16]])
-; BUDGET-NEXT:    call void @foo(i32 [[B17]])
-; BUDGET-NEXT:    ret void
-;
+define void @many_basises_overlapping(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) {
 entry:
-  %s2 = shl i32 %s, 1
-  %t1 = add i32 %b1, %s
-  call void @foo(i32 %t1)
-  %t2 = add i32 %b2, %s
-  call void @foo(i32 %t2)
-  %t3 = add i32 %b3, %s
-  call void @foo(i32 %t3)
-  %t4 = add i32 %b4, %s
-  call void @foo(i32 %t4)
-  %t5 = add i32 %b5, %s
-  call void @foo(i32 %t5)
-  %t6 = add i32 %b6, %s
-  call void @foo(i32 %t6)
-  %t7 = add i32 %b7, %s
-  call void @foo(i32 %t7)
-  %t8 = add i32 %b8, %s
-  call void @foo(i32 %t8)
-  %t9 = add i32 %b9, %s
-  call void @foo(i32 %t9)
-  %t10 = add i32 %b10, %s
-  call void @foo(i32 %t10)
-  %t11 = add i32 %b11, %s
-  call void @foo(i32 %t11)
-  %t12 = add i32 %b12, %s
-  call void @foo(i32 %t12)
-  %t13 = add i32 %b13, %s
-  call void @foo(i32 %t13)
-  %t14 = add i32 %b14, %s
-  call void @foo(i32 %t14)
-  %t15 = add i32 %b15, %s
-  call void @foo(i32 %t15)
-  %t16 = add i32 %b16, %s
-  call void @foo(i32 %t16)
-  %t17 = add i32 %b17, %s
-  call void @foo(i32 %t17)
-  %u1 = add i32 %b1, %s2
-  call void @foo(i32 %u1)
-  %u2 = add i32 %b2, %s2
-  call void @foo(i32 %u2)
-  %u3 = add i32 %b3, %s2
-  call void @foo(i32 %u3)
-  %u4 = add i32 %b4, %s2
-  call void @foo(i32 %u4)
-  %u5 = add i32 %b5, %s2
-  call void @foo(i32 %u5)
-  %u6 = add i32 %b6, %s2
-  call void @foo(i32 %u6)
-  %u7 = add i32 %b7, %s2
-  call void @foo(i32 %u7)
-  %u8 = add i32 %b8, %s2
-  call void @foo(i32 %u8)
-  %u9 = add i32 %b9, %s2
-  call void @foo(i32 %u9)
-  %u10 = add i32 %b10, %s2
-  call void @foo(i32 %u10)
-  %u11 = add i32 %b11, %s2
-  call void @foo(i32 %u11)
-  %u12 = add i32 %b12, %s2
-  call void @foo(i32 %u12)
-  %u13 = add i32 %b13, %s2
-  call void @foo(i32 %u13)
-  %u14 = add i32 %b14, %s2
-  call void @foo(i32 %u14)
-  %u15 = add i32 %b15, %s2
-  call void @foo(i32 %u15)
-  %u16 = add i32 %b16, %s2
-  call void @foo(i32 %u16)
-  %u17 = add i32 %b17, %s2
-  call void @foo(i32 %u17)
-  call void @foo(i32 %b1)
-  call void @foo(i32 %b2)
-  call void @foo(i32 %b3)
-  call void @foo(i32 %b4)
-  call void @foo(i32 %b5)
-  call void @foo(i32 %b6)
-  call void @foo(i32 %b7)
-  call void @foo(i32 %b8)
-  call void @foo(i32 %b9)
-  call void @foo(i32 %b10)
-  call void @foo(i32 %b11)
-  call void @foo(i32 %b12)
-  call void @foo(i32 %b13)
-  call void @foo(i32 %b14)
-  call void @foo(i32 %b15)
-  call void @foo(i32 %b16)
-  call void @foo(i32 %b17)
+  %s2 = shl i128 %s, 1
+  %t1 = add i128 %b1, %s
+  call void @bar(i128 %t1)
+  %t2 = add i128 %b2, %s
+  call void @bar(i128 %t2)
+  %t3 = add i128 %b3, %s
+  call void @bar(i128 %t3)
+  %t4 = add i128 %b4, %s
+  call void @bar(i128 %t4)
+  %t5 = add i128 %b5, %s
+  call void @bar(i128 %t5)
+  %t6 = add i128 %b6, %s
+  call void @bar(i128 %t6)
+  %t7 = add i128 %b7, %s
+  call void @bar(i128 %t7)
+  %t8 = add i128 %b8, %s
+  call void @bar(i128 %t8)
+  %t9 = add i128 %b9, %s
+  call void @bar(i128 %t9)
+  %t10 = add i128 %b10, %s
+  call void @bar(i128 %t10)
+  %t11 = add i128 %b11, %s
+  call void @bar(i128 %t11)
+  %t12 = add i128 %b12, %s
+  call void @bar(i128 %t12)
+  %t13 = add i128 %b13, %s
+  call void @bar(i128 %t13)
+  %t14 = add i128 %b14, %s
+  call void @bar(i128 %t14)
+  %t15 = add i128 %b15, %s
+  call void @bar(i128 %t15)
+  %t16 = add i128 %b16, %s
+  call void @bar(i128 %t16)
+  %t17 = add i128 %b17, %s
+  call void @bar(i128 %t17)
+  %u1 = add i128 %b1, %s2
+  call void @bar(i128 %u1)
+  %u2 = add i128 %b2, %s2
+  call void @bar(i128 %u2)
+  %u3 = add i128 %b3, %s2
+  call void @bar(i128 %u3)
+  %u4 = add i128 %b4, %s2
+  call void @bar(i128 %u4)
+  %u5 = add i128 %b5, %s2
+  call void @bar(i128 %u5)
+  %u6 = add i128 %b6, %s2
+  call void @bar(i128 %u6)
+  %u7 = add i128 %b7, %s2
+  call void @bar(i128 %u7)
+  %u8 = add i128 %b8, %s2
+  call void @bar(i128 %u8)
+  %u9 = add i128 %b9, %s2
+  call void @bar(i128 %u9)
+  %u10 = add i128 %b10, %s2
+  call void @bar(i128 %u10)
+  %u11 = add i128 %b11, %s2
+  call void @bar(i128 %u11)
+  %u12 = add i128 %b12, %s2
+  call void @bar(i128 %u12)
+  %u13 = add i128 %b13, %s2
+  call void @bar(i128 %u13)
+  %u14 = add i128 %b14, %s2
+  call void @bar(i128 %u14)
+  %u15 = add i128 %b15, %s2
+  call void @bar(i128 %u15)
+  %u16 = add i128 %b16, %s2
+  call void @bar(i128 %u16)
+  %u17 = add i128 %b17, %s2
+  call void @bar(i128 %u17)
+  call void @bar(i128 %b1)
+  call void @bar(i128 %b2)
+  call void @bar(i128 %b3)
+  call void @bar(i128 %b4)
+  call void @bar(i128 %b5)
+  call void @bar(i128 %b6)
+  call void @bar(i128 %b7)
+  call void @bar(i128 %b8)
+  call void @bar(i128 %b9)
+  call void @bar(i128 %b10)
+  call void @bar(i128 %b11)
+  call void @bar(i128 %b12)
+  call void @bar(i128 %b13)
+  call void @bar(i128 %b14)
+  call void @bar(i128 %b15)
+  call void @bar(i128 %b16)
+  call void @bar(i128 %b17)
   ret void
 }
-;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
-; CHECK: {{.*}}

>From 3538f0315b955cd2d39adfc147c0699045239d6e Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <yoonseo.choi at amd.com>
Date: Thu, 3 Sep 2026 00:11:22 +0000
Subject: [PATCH 09/13] per-BB threshold on num basises

---
 .../Scalar/StraightLineStrengthReduce.cpp     | 131 +++++++++---------
 ...sr-rp-filter.ll => slsr-rewrite-filter.ll} |  12 +-
 ...sr-rp-filter.ll => slsr-rewrite-filter.ll} |   7 +-
 3 files changed, 77 insertions(+), 73 deletions(-)
 rename llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/{slsr-rp-filter.ll => slsr-rewrite-filter.ll} (97%)
 rename llvm/test/Transforms/StraightLineStrengthReduce/{slsr-rp-filter.ll => slsr-rewrite-filter.ll} (92%)

diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 45dc1942e1bb7..c5e803be9cc3c 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -111,8 +111,10 @@ using namespace llvm;
 using namespace PatternMatch;
 
 #define DEBUG_TYPE "slsr"
-#define DEBUG_SLSR_RP(X) DEBUG_WITH_TYPE(DEBUG_TYPE "-rp", X)
-#define DEBUG_SLSR_RP_DETAIL(X) DEBUG_WITH_TYPE(DEBUG_TYPE "-rp-detail", X)
+#define DEBUG_SLSR_REWRITE_FILTER(X)                                           \
+  DEBUG_WITH_TYPE(DEBUG_TYPE "-rewrite-filter", X)
+#define DEBUG_SLSR_REWRITE_FILTER_DETAIL(X)                                    \
+  DEBUG_WITH_TYPE(DEBUG_TYPE "-rewrite-filter-detail", X)
 
 static const unsigned UnknownAddressSpace =
     std::numeric_limits<unsigned>::max();
@@ -125,18 +127,15 @@ static cl::opt<bool>
     EnablePoisonReuseGuard("enable-poison-reuse-guard", cl::init(true),
                            cl::desc("Enable poison-reuse guard"));
 
-static cl::opt<bool> EnableRPFilter(
-    "slsr-rp-filter", cl::init(true), cl::Hidden,
+static cl::opt<bool> EnableRewriteFilter(
+    "slsr-rewrite-filter", cl::init(true), cl::Hidden,
     cl::desc("SLSR: skip rewrites in blocks where they would push register "
              "pressure past the target's register budget"));
 
 STATISTIC(NumSCEVCandidateBasisDifferences,
           "Number of candidate-basis SCEV differences computed by SLSR");
-STATISTIC(NumRPFilteredBlocks,
-          "Number of blocks whose rewrites SLSR skipped due to register "
-          "pressure");
-STATISTIC(NumRPFilteredCandidates,
-          "Number of candidates SLSR skipped due to register pressure");
+STATISTIC(NumFilteredCandidates,
+          "Number of SLSR candidates not rewritten due to register pressure");
 
 namespace {
 
@@ -1426,28 +1425,28 @@ bool StraightLineStrengthReduceLegacyPass::runOnFunction(Function &F) {
 
 namespace {
 
-class RPFilter {
-  // RPFilter targets one pathological shape: a block holding many distinct
-  // bases. Each basis contributes one extended live range no matter how many
-  // candidates are rewritten against it, so it is the number of distinct bases,
-  // not the number of candidates, that tracks how many new concurrent live
-  // ranges SLSR would create. Below this count no block in the function can
-  // exhibit the pathology, and the liveness and pressure analyses are skipped.
-  static constexpr unsigned MinDistinctBasesToFilter = 16;
+class RewriteFilter {
+  // Recondiser rewriting a basic block holding many distinct SLSR bases.
+
+  // Each basis contributes one extended live range no matter how many
+  // candidates are rewritten against it in a basic block.
+  // If the number of bases is above this threshold do a check if the
+  // liveness of the block has increase a lot by SLSR.
+  static constexpr unsigned MinDistinctBasisesToFilter = 16;
 
 public:
   using Candidate = StraightLineStrengthReduce::Candidate;
 
-  RPFilter(const Function *F,
-           DenseMap<Instruction *, Candidate *> &PickedCandidateMap,
-           const TargetTransformInfo *TTI)
+  RewriteFilter(const Function *F,
+                DenseMap<Instruction *, Candidate *> &PickedCandidateMap,
+                const TargetTransformInfo *TTI)
       : F(F), PickedCandidateMap(PickedCandidateMap), TTI(TTI) {}
 
   // Return the candidates whose rewrite would push their block's register
   // pressure past what the target can allocate.
   DenseSet<const Instruction *> run() {
     DenseSet<const Instruction *> InstsToSkip;
-    if (!EnableRPFilter || PickedCandidateMap.empty())
+    if (!EnableRewriteFilter || PickedCandidateMap.empty())
       return InstsToSkip;
 
     // Without a budget there is nothing to compare the pressure against, so
@@ -1457,27 +1456,33 @@ class RPFilter {
       return InstsToSkip;
 
     buildBBToNumCandsAndBasises(PickedCandidateMap);
-    if (MaxNumBasisesInBB <= MinDistinctBasesToFilter)
+    if (MaxNumBasisesInBB <= MinDistinctBasisesToFilter)
       return InstsToSkip;
 
     // Compute live-in and live-out of each BB in CFG
     buildBBToLiveness(*F);
 
-    DEBUG_SLSR_RP(dbgs() << "-- MaxRP of BBs -- \n");
+    DEBUG_SLSR_REWRITE_FILTER(dbgs() << "-- MaxRP of BBs -- \n");
     SmallPtrSet<const BasicBlock *, 8> BBsToSkip;
     for (auto &BB : *F) {
+
+      auto It = BBToNumCandsAndBasises.find(&BB);
+      if (It == BBToNumCandsAndBasises.end() ||
+          It->second.second <= MinDistinctBasisesToFilter)
+        continue;
+
       const BlockLiveness &BL = getLiveness(&BB);
       auto [MaxRP, MaxRPWithSLSR] =
-          maxPressureInBlockBackward(BB, BL.LiveIn, BL.LiveOut);
-      DEBUG_SLSR_RP(dbgs() << "MaxRP:" << BB.getName() << ": (" << MaxRP << ", "
-                           << MaxRPWithSLSR << ")" << "\n");
+          maxLivenessInBlockBackward(BB, BL.LiveIn, BL.LiveOut);
+      DEBUG_SLSR_REWRITE_FILTER(dbgs()
+                                << "MaxRP:" << BB.getName() << ": (" << MaxRP
+                                << ", " << MaxRPWithSLSR << ")" << "\n");
 
       if (!rewriteWouldOverflowBudget(MaxRP, MaxRPWithSLSR, *Budget))
         continue;
 
-      DEBUG_SLSR_RP(dbgs() << "Skipping BB from SLSR: " << BB.getName()
-                           << "\n");
-      ++NumRPFilteredBlocks;
+      DEBUG_SLSR_REWRITE_FILTER(
+          dbgs() << "Skipping BB from SLSR: " << BB.getName() << "\n");
       BBsToSkip.insert(&BB);
     } // Done with BBs
 
@@ -1487,7 +1492,7 @@ class RPFilter {
     for (const auto &It : PickedCandidateMap)
       if (BBsToSkip.contains(It.first->getParent()) &&
           InstsToSkip.insert(It.first).second)
-        ++NumRPFilteredCandidates;
+        NumFilteredCandidates++;
 
     return InstsToSkip;
   }
@@ -1508,6 +1513,9 @@ class RPFilter {
   };
 
   DenseMap<const BasicBlock *, BlockLiveness> BBToLiveness;
+  // The liveness scan asks for the weight of every live value at every
+  // instruction, so memoize on the type, which is all the weight depends on.
+  mutable DenseMap<Type *, unsigned> WeightCache;
 
   // Liveness is only computed for functions that pass the candidate-count
   // gate. Blocks of the remaining functions read as having nothing live across
@@ -1563,7 +1571,7 @@ class RPFilter {
       }
     }
     MaxNumBasisesInBB = std::max(MaxNumBasisesInBB, UniqueBasises.size());
-    DEBUG_WITH_TYPE("slsr-rp", {
+    DEBUG_SLSR_REWRITE_FILTER({
       dbgs() << "BB: " << BB->getName() << " - NumCands: " << NumCands
              << " UniqueBasises: " << UniqueBasises.size() << "\n";
     });
@@ -1584,29 +1592,25 @@ class RPFilter {
     return isa<Instruction>(V) || isa<Argument>(V);
   }
 
-  // The pressure scan asks for the weight of every live value at every
-  // instruction, so memoize on the type, which is all the weight depends on.
-  mutable DenseMap<Type *, unsigned> RegWeightCache;
+  unsigned computeWeight(Type *Ty) const {
+    const DataLayout &DL = F->getDataLayout();
+
+    // TTI's getRegUsageForType is less accurate than
+    // default logic to compute RP for targets like AMDGPU
+    return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
+  }
 
-  unsigned regWeight(const Value *V) const {
+  unsigned weight(const Value *V) const {
     Type *Ty = V->getType();
     if (Ty->isVoidTy() || Ty->isTokenTy())
       return 0;
 
-    auto [It, Inserted] = RegWeightCache.try_emplace(Ty);
+    auto [It, Inserted] = WeightCache.try_emplace(Ty);
     if (Inserted)
-      It->second = computeRegWeight(Ty);
+      It->second = computeWeight(Ty);
     return It->second;
   }
 
-  unsigned computeRegWeight(Type *Ty) const {
-    const DataLayout &DL = F->getDataLayout();
-
-    // TTI's getRegUsageForType is less accurate than
-    // default logic to compute RP for targets like AMDGPU
-    return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
-  }
-
   void buildBBToLiveness(const Function &F) {
     DenseMap<const BasicBlock *, ValueSet> UpExposed, Defs;
     DenseMap<std::pair<const BasicBlock *, const BasicBlock *>, ValueSet>
@@ -1632,7 +1636,7 @@ class RPFilter {
         if (!I.getType()->isVoidTy())
           D.insert(&I);
       }
-      DEBUG_SLSR_RP_DETAIL({
+      DEBUG_SLSR_REWRITE_FILTER_DETAIL({
         dbgs() << "BB-fill: " << BB.getName()
                << " UpExposed: " << UpExposed[&BB].size();
         dbgs() << " Defs: " << Defs[&BB].size() << "\n";
@@ -1665,7 +1669,7 @@ class RPFilter {
             Out.size() != BBToLiveness[BB].LiveOut.size())
           Changed = true;
 
-        DEBUG_SLSR_RP_DETAIL({
+        DEBUG_SLSR_REWRITE_FILTER_DETAIL({
           dbgs() << "BB-update: " << BB->getName() << " In: " << In.size()
                  << " Out: " << Out.size() << "\n";
         });
@@ -1675,7 +1679,7 @@ class RPFilter {
       }
     }
 
-    DEBUG_WITH_TYPE("slsr-rp", {
+    DEBUG_WITH_TYPE("slsr-rewrite-filter", {
       dbgs() << "-- Live Ins/Outs of BBs -- \n";
       for (const BasicBlock &BB : F) {
         dbgs() << BB.getName() << ": ";
@@ -1712,15 +1716,13 @@ class RPFilter {
   }
 
   std::pair<unsigned, unsigned>
-  maxPressureInBlockBackward(const BasicBlock &BB, const ValueSet &LiveIn,
+  maxLivenessInBlockBackward(const BasicBlock &BB, const ValueSet &LiveIn,
                              const ValueSet &LiveOut) const {
 
-    // TODO: ValueSet is a SmallPtrSet, which is supposedly smaller than 33.
-    //       Could be a better-fitting data structure.
     ValueSet LiveSet = LiveOut;
     unsigned MaxW = 0;
     for (const Value *V : LiveSet)
-      MaxW += regWeight(V);
+      MaxW += weight(V);
 
     // Initial LiveSetWithSLSR is the same as LiveOut.
     // SLSR changes
@@ -1754,10 +1756,10 @@ class RPFilter {
             // removed from LiveSetWithSLSR. As a heuristic, we don't dicern
             // which Op of I is replaced by Basis. When IsCand is true, there is
             // only one reg-like Op in practice.
-            // TODO: Can further restrict by Candidate's type and DeltaKind.
             LiveSetWithSLSR.erase(Op);
-            DEBUG_SLSR_RP_DETAIL(dbgs() << "Removed Op from LiveSetWithSLSR: "
-                                        << *Op << " in inst " << I << "\n");
+            DEBUG_SLSR_REWRITE_FILTER_DETAIL(
+                dbgs() << "Removed Op from LiveSetWithSLSR: " << *Op
+                       << " in inst " << I << "\n");
           } else {
             LiveSetWithSLSR.insert(Op);
           }
@@ -1770,13 +1772,13 @@ class RPFilter {
       // Compute RP reaching for this instruction.
       unsigned W = 0;
       for (const Value *V : LiveSet) {
-        auto RW = regWeight(V);
+        auto RW = weight(V);
         W += RW;
       }
       MaxW = std::max(MaxW, W);
       W = 0;
       for (const Value *V : LiveSetWithSLSR) {
-        auto RW = regWeight(V);
+        auto RW = weight(V);
         W += RW;
       }
       MaxWWithSLSR = std::max(MaxWWithSLSR, W);
@@ -1810,6 +1812,7 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
   }
   sortCandidateInstructions();
 
+  // Keep picked candidates not to call pickRewriteCandidate() again.
   DenseMap<Instruction *, Candidate *> PickedCandidateMap;
   for (Instruction *I : SortedCandidateInsts)
     if (Candidate *C = pickRewriteCandidate(I))
@@ -1818,18 +1821,18 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
   // Candidates whose rewrite would push their block's register pressure past
   // what the target can allocate. Evaluated on the original IR, before any
   // rewriteCandidate mutates it: rewriting inserts instructions and calls
-  // replaceAllUsesWith, which would invalidate the liveness and pressure
+  // replaceAllUsesWith, which would invalidate the liveness
   // analyses the filter relies on.
-  RPFilter RPFilter(&F, PickedCandidateMap, TTI);
-  DenseSet<const Instruction *> ToSkipRewrite = RPFilter.run();
+  RewriteFilter RewriteFilter(&F, PickedCandidateMap, TTI);
+  DenseSet<const Instruction *> ToSkipRewrite = RewriteFilter.run();
 
   // Rewrite candidates in the topological order that rewrites a Candidate
   // always before rewriting its Basis
   for (Instruction *I : reverse(SortedCandidateInsts)) {
-    if (ToSkipRewrite.contains(I))
-      continue;
-    if (Candidate *C = pickRewriteCandidate(I))
-      rewriteCandidate(*C);
+    auto It = PickedCandidateMap.find(I);
+    if (It != PickedCandidateMap.end())
+      if (!ToSkipRewrite.contains(It->first))
+        rewriteCandidate(*It->second);
   }
 
   for (auto *DeadIns : DeadInstructions)
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter.ll
similarity index 97%
rename from llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
rename to llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter.ll
index a5f86a1a42ce4..efcc3996a7ead 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rp-filter.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter.ll
@@ -2,9 +2,9 @@
 ; RUN: opt < %s -mtriple=amdgpu-amdhsa-amd -passes=slsr -S | FileCheck %s
 
 ; SLSR rewrites a candidate as its basis plus a delta, which keeps the basis live
-; down to the candidate. The register-pressure filter drops a block's rewrites
-; when doing so would take the block's peak pressure past what the target can
-; allocate.
+; down to the candidate. (Multiple candidates can be rewritten off of one basis). 
+; If the rewrites are likely to increase the basic block's peak pressure what 
+; the target can allocate, the rewrites are skipped. 
 ;
 ; The budget comes from TTI, which on AMDGPU reports the VGPR count implied by
 ; the occupancy the function is compiled for. This triple selects the generic
@@ -14,8 +14,8 @@
 ;
 ; Both functions below have byte-identical bodies: 17 distinct bases whose
 ; rewrites take the block's peak pressure from 80 registers to 144. The only
-; difference is the occupancy attribute, so the budget is the only variable and
-; the comparison isolates it.
+; difference is the occupancy attribute to demonstrate the difference with 
+; different budgets
 ;
 ;   @peak_above_budget  no attribute. The work group size defaults to 1024
 ;                       threads, which is 16 wave64s over 4 EUs, so 4 waves per
@@ -391,4 +391,4 @@ entry:
   ret void
 }
 
-attributes #0 = { "amdgpu-flat-work-group-size"="1,256" }
\ No newline at end of file
+attributes #0 = { "amdgpu-flat-work-group-size"="1,256" }
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
similarity index 92%
rename from llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
rename to llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
index 32761d2922acd..fcc38a8dd43d5 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rp-filter.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
@@ -1,16 +1,17 @@
+; NOTE: Do not auto-generate
 ; REQUIRES: asserts
 ; RUN: opt -passes=slsr -stats -disable-output <%s 2>&1 | FileCheck %s
 
-; CHECK-NOT: Number of blocks whose rewrites SLSR skipped due to register pressure
+; CHECK-NOT: Number of SLSR candidates not rewritten due to register pressure
 
-; The register-pressure filter needs a register budget to compare against, and
+; The SLSR rewirte filter needs a register budget to compare against, and
 ; the generic TargetTransformInfo has none: getRegisterBudget() returns
 ; std::nullopt unless a target implements it. There is no target triple here, so
 ; RPFilter::run() returns early, before it even computes liveness or pressure,
 ; and every rewrite stands.
 ;
 ; @many_basises_overlapping is the same body used by
-; @peak_above_budget in AMDGPU/slsr-rp-filter.ll: 17 distinct basises whose
+; @peak_above_budget in AMDGPU/slsr-rewrite-filter.ll: 17 distinct basises whose
 ; rewrites take the block's peak pressure from 80 registers to 144. Under an
 ; AMDGPU triple that exceeds the budget and the rewrites are dropped.
 ;

>From b0e285cea6bdf7cf9a65be169c1125e33537d9fa Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <Yoonseo.Choi at amd.com>
Date: Mon, 14 Sep 2026 18:36:07 -0500
Subject: [PATCH 10/13] Formatting

---
 .../Scalar/StraightLineStrengthReduce.cpp     | 25 ++++++++++---------
 1 file changed, 13 insertions(+), 12 deletions(-)

diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index c5e803be9cc3c..4ad67ac907f16 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -1462,7 +1462,7 @@ class RewriteFilter {
     // Compute live-in and live-out of each BB in CFG
     buildBBToLiveness(*F);
 
-    DEBUG_SLSR_REWRITE_FILTER(dbgs() << "-- MaxRP of BBs -- \n");
+    DEBUG_SLSR_REWRITE_FILTER(dbgs() << "-- Max liveness of BBs -- \n");
     SmallPtrSet<const BasicBlock *, 8> BBsToSkip;
     for (auto &BB : *F) {
 
@@ -1472,13 +1472,14 @@ class RewriteFilter {
         continue;
 
       const BlockLiveness &BL = getLiveness(&BB);
-      auto [MaxRP, MaxRPWithSLSR] =
+      auto [MaxLiveness, MaxLivenessWithSLSR] =
           maxLivenessInBlockBackward(BB, BL.LiveIn, BL.LiveOut);
-      DEBUG_SLSR_REWRITE_FILTER(dbgs()
-                                << "MaxRP:" << BB.getName() << ": (" << MaxRP
-                                << ", " << MaxRPWithSLSR << ")" << "\n");
+      DEBUG_SLSR_REWRITE_FILTER(dbgs() << "MaxLiveness:" << BB.getName()
+                                       << ": (" << MaxLiveness << ", "
+                                       << MaxLivenessWithSLSR << ")" << "\n");
 
-      if (!rewriteWouldOverflowBudget(MaxRP, MaxRPWithSLSR, *Budget))
+      if (!rewriteWouldOverflowBudget(MaxLiveness, MaxLivenessWithSLSR,
+                                      *Budget))
         continue;
 
       DEBUG_SLSR_REWRITE_FILTER(
@@ -1540,8 +1541,8 @@ class RewriteFilter {
                                   unsigned Budget) const {
     // Leave the allocator some slack: it also has to satisfy register class
     // and ABI constraints that this estimate knows nothing about.
-    constexpr double SLSRRPSafeFraction = 0.9;
-    unsigned SafeBudget = static_cast<unsigned>(Budget * SLSRRPSafeFraction);
+    constexpr double SafeRatio = 0.9;
+    unsigned SafeBudget = static_cast<unsigned>(Budget * SafeRatio);
 
     // There is headroom, so however much the rewrite adds is irrelevant.
     if (After <= SafeBudget)
@@ -1553,8 +1554,8 @@ class RewriteFilter {
 
     // Already over budget. SLSR can still lower pressure here, so only refuse
     // rewrites that make it meaningfully worse.
-    constexpr unsigned SLSRRPAbsDelta = 4;
-    return After > Before && After - Before > SLSRRPAbsDelta;
+    constexpr unsigned AbsDelta = 4;
+    return After > Before && After - Before > AbsDelta;
   }
 
   std::pair<unsigned, unsigned> countCandsAndBasisesInBB(
@@ -1596,7 +1597,7 @@ class RewriteFilter {
     const DataLayout &DL = F->getDataLayout();
 
     // TTI's getRegUsageForType is less accurate than
-    // default logic to compute RP for targets like AMDGPU
+    // default logic to compute pressure for targets like AMDGPU
     return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
   }
 
@@ -1769,7 +1770,7 @@ class RewriteFilter {
           SeenLastUse.insert(Op);
         }
 
-      // Compute RP reaching for this instruction.
+      // Compute weight reaching for this instruction.
       unsigned W = 0;
       for (const Value *V : LiveSet) {
         auto RW = weight(V);

>From aea1cc9687fc641b01b987db58d6e54b156b7bf6 Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <Yoonseo.Choi at amd.com>
Date: Wed, 23 Sep 2026 21:14:05 -0500
Subject: [PATCH 11/13] Update logic for a safe budget. Fixed some typos.

---
 .../Scalar/StraightLineStrengthReduce.cpp         | 15 +++++++--------
 .../slsr-rewrite-filter.ll                        |  4 ++--
 2 files changed, 9 insertions(+), 10 deletions(-)

diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 4ad67ac907f16..20b5376fca492 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -136,6 +136,7 @@ STATISTIC(NumSCEVCandidateBasisDifferences,
           "Number of candidate-basis SCEV differences computed by SLSR");
 STATISTIC(NumFilteredCandidates,
           "Number of SLSR candidates not rewritten due to register pressure");
+STATISTIC(NumRewrittenCandidates, "Number of SLSR candidates rewritten");
 
 namespace {
 
@@ -1426,7 +1427,7 @@ bool StraightLineStrengthReduceLegacyPass::runOnFunction(Function &F) {
 namespace {
 
 class RewriteFilter {
-  // Recondiser rewriting a basic block holding many distinct SLSR bases.
+  // Reconsider rewriting a basic block holding many distinct SLSR bases.
 
   // Each basis contributes one extended live range no matter how many
   // candidates are rewritten against it in a basic block.
@@ -1465,7 +1466,6 @@ class RewriteFilter {
     DEBUG_SLSR_REWRITE_FILTER(dbgs() << "-- Max liveness of BBs -- \n");
     SmallPtrSet<const BasicBlock *, 8> BBsToSkip;
     for (auto &BB : *F) {
-
       auto It = BBToNumCandsAndBasises.find(&BB);
       if (It == BBToNumCandsAndBasises.end() ||
           It->second.second <= MinDistinctBasisesToFilter)
@@ -1518,9 +1518,6 @@ class RewriteFilter {
   // instruction, so memoize on the type, which is all the weight depends on.
   mutable DenseMap<Type *, unsigned> WeightCache;
 
-  // Liveness is only computed for functions that pass the candidate-count
-  // gate. Blocks of the remaining functions read as having nothing live across
-  // their boundaries, which is why no rewrite is suppressed in that case.
   const BlockLiveness &getLiveness(const BasicBlock *BB) const {
     static const BlockLiveness Empty;
     auto It = BBToLiveness.find(BB);
@@ -1541,8 +1538,8 @@ class RewriteFilter {
                                   unsigned Budget) const {
     // Leave the allocator some slack: it also has to satisfy register class
     // and ABI constraints that this estimate knows nothing about.
-    constexpr double SafeRatio = 0.9;
-    unsigned SafeBudget = static_cast<unsigned>(Budget * SafeRatio);
+    constexpr unsigned SafeMargin = 8;
+    unsigned SafeBudget = Budget >= SafeMargin ? Budget - SafeMargin : Budget;
 
     // There is headroom, so however much the rewrite adds is irrelevant.
     if (After <= SafeBudget)
@@ -1832,8 +1829,10 @@ bool StraightLineStrengthReduce::runOnFunction(Function &F) {
   for (Instruction *I : reverse(SortedCandidateInsts)) {
     auto It = PickedCandidateMap.find(I);
     if (It != PickedCandidateMap.end())
-      if (!ToSkipRewrite.contains(It->first))
+      if (!ToSkipRewrite.contains(It->first)) {
         rewriteCandidate(*It->second);
+        NumRewrittenCandidates++;
+      }
   }
 
   for (auto *DeadIns : DeadInstructions)
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
index fcc38a8dd43d5..66379fd2b01c0 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
@@ -4,10 +4,10 @@
 
 ; CHECK-NOT: Number of SLSR candidates not rewritten due to register pressure
 
-; The SLSR rewirte filter needs a register budget to compare against, and
+; The SLSR rewrite filter needs a register budget to compare against, and
 ; the generic TargetTransformInfo has none: getRegisterBudget() returns
 ; std::nullopt unless a target implements it. There is no target triple here, so
-; RPFilter::run() returns early, before it even computes liveness or pressure,
+; RewriteFilter::run() returns early, before it even computes liveness or pressure,
 ; and every rewrite stands.
 ;
 ; @many_basises_overlapping is the same body used by

>From e921b67607b2ed38bc001ac35c5aed5a1e38472a Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <Yoonseo.Choi at amd.com>
Date: Fri, 25 Sep 2026 11:26:08 -0500
Subject: [PATCH 12/13] Bail out on Scalable type; Remove CHECK-NOT

---
 .../Scalar/StraightLineStrengthReduce.cpp     |  69 ++++++++----
 .../AMDGPU/slsr-rewrite-filter-scalable.ll    | 103 ++++++++++++++++++
 .../AMDGPU/slsr-rewrite-filter.ll             |   9 ++
 .../slsr-rewrite-filter.ll                    |  13 +--
 4 files changed, 164 insertions(+), 30 deletions(-)
 create mode 100644 llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter-scalable.ll

diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index 20b5376fca492..d09b5c8da9e88 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -1472,8 +1472,13 @@ class RewriteFilter {
         continue;
 
       const BlockLiveness &BL = getLiveness(&BB);
-      auto [MaxLiveness, MaxLivenessWithSLSR] =
-          maxLivenessInBlockBackward(BB, BL.LiveIn, BL.LiveOut);
+      auto Liveness = maxLivenessInBlockBackward(BB, BL.LiveIn, BL.LiveOut);
+      // A block holding a value with no fixed register footprint cannot be
+      // weighed against the budget, so leave its rewrites alone.
+      if (!Liveness)
+        continue;
+
+      auto [MaxLiveness, MaxLivenessWithSLSR] = *Liveness;
       DEBUG_SLSR_REWRITE_FILTER(dbgs() << "MaxLiveness:" << BB.getName()
                                        << ": (" << MaxLiveness << ", "
                                        << MaxLivenessWithSLSR << ")" << "\n");
@@ -1516,7 +1521,7 @@ class RewriteFilter {
   DenseMap<const BasicBlock *, BlockLiveness> BBToLiveness;
   // The liveness scan asks for the weight of every live value at every
   // instruction, so memoize on the type, which is all the weight depends on.
-  mutable DenseMap<Type *, unsigned> WeightCache;
+  mutable DenseMap<Type *, std::optional<unsigned>> WeightCache;
 
   const BlockLiveness &getLiveness(const BasicBlock *BB) const {
     static const BlockLiveness Empty;
@@ -1590,15 +1595,23 @@ class RewriteFilter {
     return isa<Instruction>(V) || isa<Argument>(V);
   }
 
-  unsigned computeWeight(Type *Ty) const {
+  // Returns std::nullopt for a type with no fixed register footprint, which is
+  // the signal to stop weighing the block.
+  std::optional<unsigned> computeWeight(Type *Ty) const {
     const DataLayout &DL = F->getDataLayout();
 
+    // A scalable type occupies vscale times its minimum size, which is a
+    // runtime quantity, so it cannot be compared against a fixed budget.
+    TypeSize Size = DL.getTypeSizeInBits(Ty);
+    if (Size.isScalable())
+      return std::nullopt;
+
     // TTI's getRegUsageForType is less accurate than
     // default logic to compute pressure for targets like AMDGPU
-    return divideCeil(DL.getTypeSizeInBits(Ty).getFixedValue(), 32);
+    return divideCeil(Size.getFixedValue(), 32);
   }
 
-  unsigned weight(const Value *V) const {
+  std::optional<unsigned> weight(const Value *V) const {
     Type *Ty = V->getType();
     if (Ty->isVoidTy() || Ty->isTokenTy())
       return 0;
@@ -1713,14 +1726,27 @@ class RewriteFilter {
     return false;
   }
 
-  std::pair<unsigned, unsigned>
+  // Returns std::nullopt when a live value has no fixed register footprint, so
+  // the block's pressure cannot be compared against the budget.
+  std::optional<std::pair<unsigned, unsigned>>
   maxLivenessInBlockBackward(const BasicBlock &BB, const ValueSet &LiveIn,
                              const ValueSet &LiveOut) const {
+    auto SumWeights = [this](const ValueSet &Live) -> std::optional<unsigned> {
+      unsigned W = 0;
+      for (const Value *V : Live) {
+        std::optional<unsigned> RW = weight(V);
+        if (!RW)
+          return std::nullopt;
+        W += *RW;
+      }
+      return W;
+    };
 
     ValueSet LiveSet = LiveOut;
-    unsigned MaxW = 0;
-    for (const Value *V : LiveSet)
-      MaxW += weight(V);
+    std::optional<unsigned> InitialW = SumWeights(LiveSet);
+    if (!InitialW)
+      return std::nullopt;
+    unsigned MaxW = *InitialW;
 
     // Initial LiveSetWithSLSR is the same as LiveOut.
     // SLSR changes
@@ -1768,18 +1794,15 @@ class RewriteFilter {
         }
 
       // Compute weight reaching for this instruction.
-      unsigned W = 0;
-      for (const Value *V : LiveSet) {
-        auto RW = weight(V);
-        W += RW;
-      }
-      MaxW = std::max(MaxW, W);
-      W = 0;
-      for (const Value *V : LiveSetWithSLSR) {
-        auto RW = weight(V);
-        W += RW;
-      }
-      MaxWWithSLSR = std::max(MaxWWithSLSR, W);
+      std::optional<unsigned> W = SumWeights(LiveSet);
+      if (!W)
+        return std::nullopt;
+      std::optional<unsigned> WWithSLSR = SumWeights(LiveSetWithSLSR);
+      if (!WWithSLSR)
+        return std::nullopt;
+
+      MaxW = std::max(MaxW, *W);
+      MaxWWithSLSR = std::max(MaxWWithSLSR, *WWithSLSR);
 
       // Remove the def (Instruction) from LiveSet for the next upward
       // instruction.
@@ -1789,7 +1812,7 @@ class RewriteFilter {
       }
     }
 
-    return {MaxW, MaxWWithSLSR};
+    return {{MaxW, MaxWWithSLSR}};
   }
 };
 } // end of anonymous namespace
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter-scalable.ll b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter-scalable.ll
new file mode 100644
index 0000000000000..0c4be16d37232
--- /dev/null
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter-scalable.ll
@@ -0,0 +1,103 @@
+; NOTE: Do not auto-generate
+; REQUIRES: asserts
+; RUN: opt < %s -mtriple=amdgpu-amdhsa-amd -passes=slsr -stats -disable-output 2>&1 | FileCheck %s
+
+; The body is @peak_above_budget from slsr-rewrite-filter.ll, which is filtered
+; under this triple, plus one scalable value live across the block. All 17
+; candidates being rewritten. SLSR's RewriteFilter bails out on scalable types. 
+; CHECK: {{^ *}}17 slsr{{ +}}- Number of SLSR candidates rewritten
+
+declare void @bar(i128)
+declare void @vec_use(<vscale x 4 x i32>)
+
+define void @scalable_live_value(<vscale x 4 x i32> %v, i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) {
+entry:
+  %s2 = shl i128 %s, 1
+  %t1 = add i128 %b1, %s
+  call void @bar(i128 %t1)
+  %t2 = add i128 %b2, %s
+  call void @bar(i128 %t2)
+  %t3 = add i128 %b3, %s
+  call void @bar(i128 %t3)
+  %t4 = add i128 %b4, %s
+  call void @bar(i128 %t4)
+  %t5 = add i128 %b5, %s
+  call void @bar(i128 %t5)
+  %t6 = add i128 %b6, %s
+  call void @bar(i128 %t6)
+  %t7 = add i128 %b7, %s
+  call void @bar(i128 %t7)
+  %t8 = add i128 %b8, %s
+  call void @bar(i128 %t8)
+  %t9 = add i128 %b9, %s
+  call void @bar(i128 %t9)
+  %t10 = add i128 %b10, %s
+  call void @bar(i128 %t10)
+  %t11 = add i128 %b11, %s
+  call void @bar(i128 %t11)
+  %t12 = add i128 %b12, %s
+  call void @bar(i128 %t12)
+  %t13 = add i128 %b13, %s
+  call void @bar(i128 %t13)
+  %t14 = add i128 %b14, %s
+  call void @bar(i128 %t14)
+  %t15 = add i128 %b15, %s
+  call void @bar(i128 %t15)
+  %t16 = add i128 %b16, %s
+  call void @bar(i128 %t16)
+  %t17 = add i128 %b17, %s
+  call void @bar(i128 %t17)
+  %u1 = add i128 %b1, %s2
+  call void @bar(i128 %u1)
+  %u2 = add i128 %b2, %s2
+  call void @bar(i128 %u2)
+  %u3 = add i128 %b3, %s2
+  call void @bar(i128 %u3)
+  %u4 = add i128 %b4, %s2
+  call void @bar(i128 %u4)
+  %u5 = add i128 %b5, %s2
+  call void @bar(i128 %u5)
+  %u6 = add i128 %b6, %s2
+  call void @bar(i128 %u6)
+  %u7 = add i128 %b7, %s2
+  call void @bar(i128 %u7)
+  %u8 = add i128 %b8, %s2
+  call void @bar(i128 %u8)
+  %u9 = add i128 %b9, %s2
+  call void @bar(i128 %u9)
+  %u10 = add i128 %b10, %s2
+  call void @bar(i128 %u10)
+  %u11 = add i128 %b11, %s2
+  call void @bar(i128 %u11)
+  %u12 = add i128 %b12, %s2
+  call void @bar(i128 %u12)
+  %u13 = add i128 %b13, %s2
+  call void @bar(i128 %u13)
+  %u14 = add i128 %b14, %s2
+  call void @bar(i128 %u14)
+  %u15 = add i128 %b15, %s2
+  call void @bar(i128 %u15)
+  %u16 = add i128 %b16, %s2
+  call void @bar(i128 %u16)
+  %u17 = add i128 %b17, %s2
+  call void @bar(i128 %u17)
+  call void @bar(i128 %b1)
+  call void @bar(i128 %b2)
+  call void @bar(i128 %b3)
+  call void @bar(i128 %b4)
+  call void @bar(i128 %b5)
+  call void @bar(i128 %b6)
+  call void @bar(i128 %b7)
+  call void @bar(i128 %b8)
+  call void @bar(i128 %b9)
+  call void @bar(i128 %b10)
+  call void @bar(i128 %b11)
+  call void @bar(i128 %b12)
+  call void @bar(i128 %b13)
+  call void @bar(i128 %b14)
+  call void @bar(i128 %b15)
+  call void @bar(i128 %b16)
+  call void @bar(i128 %b17)
+  call void @vec_use(<vscale x 4 x i32> %v)
+  ret void
+}
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter.ll
index efcc3996a7ead..49e80a3b64615 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/AMDGPU/slsr-rewrite-filter.ll
@@ -1,5 +1,14 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; RUN: opt < %s -mtriple=amdgpu-amdhsa-amd -passes=slsr -S | FileCheck %s
+; REQUIRES: asserts
+; RUN: opt < %s -mtriple=amdgpu-amdhsa-amd -passes=slsr -stats -disable-output 2>&1 \
+; RUN:   | FileCheck %s --check-prefix=STATS
+
+; Statistics are module wide, so the two counters below separate the two
+; functions: the 17 skipped candidates come from @peak_above_budget and the 17
+; rewritten ones from @peak_below_budget. 
+; STATS: {{^ *}}17 slsr{{ +}}- Number of SLSR candidates not rewritten due to register pressure
+; STATS: {{^ *}}17 slsr{{ +}}- Number of SLSR candidates rewritten
 
 ; SLSR rewrites a candidate as its basis plus a delta, which keeps the basis live
 ; down to the candidate. (Multiple candidates can be rewritten off of one basis). 
diff --git a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
index 66379fd2b01c0..d1609bc392326 100644
--- a/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
+++ b/llvm/test/Transforms/StraightLineStrengthReduce/slsr-rewrite-filter.ll
@@ -2,7 +2,7 @@
 ; REQUIRES: asserts
 ; RUN: opt -passes=slsr -stats -disable-output <%s 2>&1 | FileCheck %s
 
-; CHECK-NOT: Number of SLSR candidates not rewritten due to register pressure
+; CHECK: {{^ *}}17 slsr{{ +}}- Number of SLSR candidates rewritten
 
 ; The SLSR rewrite filter needs a register budget to compare against, and
 ; the generic TargetTransformInfo has none: getRegisterBudget() returns
@@ -10,20 +10,19 @@
 ; RewriteFilter::run() returns early, before it even computes liveness or pressure,
 ; and every rewrite stands.
 ;
-; @many_basises_overlapping is the same body used by
-; @peak_above_budget in AMDGPU/slsr-rewrite-filter.ll: 17 distinct basises whose
+; @many_bases_overlapping is the same body used by
+; @peak_above_budget in AMDGPU/slsr-rewrite-filter.ll: 17 distinct bases whose
 ; rewrites take the block's peak pressure from 80 registers to 144. Under an
 ; AMDGPU triple that exceeds the budget and the rewrites are dropped.
 ;
-; The check is on the statistic rather than on the IR because LLVM only prints a
-; statistic that was incremented at least once, so a missing counter means the
-; filter never skipped a block.
+; All 17 candidates being rewritten is what shows the filter never skipped a
+; block.
 
 target datalayout = "e-i64:64-v16:16-v32:32-n16:32:64"
 
 declare void @bar(i128)
 
-define void @many_basises_overlapping(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) {
+define void @many_bases_overlapping(i128 %s, i128 %b1, i128 %b2, i128 %b3, i128 %b4, i128 %b5, i128 %b6, i128 %b7, i128 %b8, i128 %b9, i128 %b10, i128 %b11, i128 %b12, i128 %b13, i128 %b14, i128 %b15, i128 %b16, i128 %b17) {
 entry:
   %s2 = shl i128 %s, 1
   %t1 = add i128 %b1, %s

>From 595f4be5526e9efa9746715121a15af1c9d2e48a Mon Sep 17 00:00:00 2001
From: Yoonseo Choi <Yoonseo.Choi at amd.com>
Date: Thu, 1 Oct 2026 22:22:09 -0500
Subject: [PATCH 13/13] Move AMDGPU specific logics into TTI interface; clarify
 comments

---
 .../llvm/Analysis/TargetTransformInfo.h       | 19 +++++++++++--------
 .../AMDGPU/AMDGPUTargetTransformInfo.cpp      |  6 +++++-
 .../Scalar/StraightLineStrengthReduce.cpp     | 18 +++++++-----------
 3 files changed, 23 insertions(+), 20 deletions(-)

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index e1d6c692f1c3b..413b0f126acac 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -1359,15 +1359,18 @@ class TargetTransformInfo {
   /// \return the number of registers in the target-provided register class.
   LLVM_ABI unsigned getNumberOfRegisters(unsigned ClassID) const;
 
-  /// \return The number of registers available to \p F before the register
-  /// allocator is forced to spill, or std::nullopt if the target cannot
-  /// provide a meaningful bound.
+  /// \return A conservative bound on the number of registers \p F can use
+  /// before the register allocator is likely to spill, or std::nullopt if the
+  /// target has no meaningful bound to report.
   ///
-  /// Unlike getNumberOfRegisters(), which some targets deliberately
-  /// under-report to tune vectorization and interleaving, this is meant to be
-  /// a real budget usable by register-pressure heuristics. Targets with
-  /// several register files report the budget for the file that dominates
-  /// pressure.
+  /// The budget is a property of the function, not only of the subtarget: it
+  /// may depend on attributes that constrain how many registers the function
+  /// is permitted to use. So two functions in the same module can have
+  /// different budgets.
+  ///
+  /// It is intended for heuristics deciding whether a
+  /// transform is about to make register pressure a problem, and staying below.
+  /// Not a guarantee that no spilling occurs.
   LLVM_ABI std::optional<unsigned> getRegisterBudget(const Function &F) const;
 
   /// \return true if the target supports load/store that enables fault
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index 16555675edda3..8c0a0897508ef 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -313,7 +313,11 @@ std::optional<unsigned> GCNTTIImpl::getRegisterBudget(const Function &F) const {
   // Report the VGPR budget implied by the occupancy F is compiled for. Callers
   // comparing a single lumped pressure number against this should be
   // conservative on the SGPR side, which is intentional.
-  return ST->getMaxNumVGPRs(F);
+  // Leave the allocator some slack.
+  constexpr unsigned SafeMargin = 8;
+  unsigned Budget = ST->getMaxNumVGPRs(F);
+  unsigned SafeBudget = Budget >= SafeMargin ? Budget - SafeMargin : Budget;
+  return SafeBudget;
 }
 
 TypeSize
diff --git a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
index d09b5c8da9e88..3fce004e75b97 100644
--- a/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/StraightLineStrengthReduce.cpp
@@ -1541,23 +1541,16 @@ class RewriteFilter {
   // registers than \p Budget.
   bool rewriteWouldOverflowBudget(unsigned Before, unsigned After,
                                   unsigned Budget) const {
-    // Leave the allocator some slack: it also has to satisfy register class
-    // and ABI constraints that this estimate knows nothing about.
-    constexpr unsigned SafeMargin = 8;
-    unsigned SafeBudget = Budget >= SafeMargin ? Budget - SafeMargin : Budget;
 
     // There is headroom, so however much the rewrite adds is irrelevant.
-    if (After <= SafeBudget)
+    if (After <= Budget)
       return false;
 
     // The rewrite is what takes the block over.
-    if (Before <= SafeBudget)
+    if (Before <= Budget)
       return true;
 
-    // Already over budget. SLSR can still lower pressure here, so only refuse
-    // rewrites that make it meaningfully worse.
-    constexpr unsigned AbsDelta = 4;
-    return After > Before && After - Before > AbsDelta;
+    return After > Before && After - Before > 0;
   }
 
   std::pair<unsigned, unsigned> countCandsAndBasisesInBB(
@@ -1608,7 +1601,10 @@ class RewriteFilter {
 
     // TTI's getRegUsageForType is less accurate than
     // default logic to compute pressure for targets like AMDGPU
-    return divideCeil(Size.getFixedValue(), 32);
+    unsigned RegisterBitWidth =
+        TTI->getRegisterBitWidth(TargetTransformInfo::RGK_Scalar)
+            .getFixedValue();
+    return divideCeil(Size.getFixedValue(), RegisterBitWidth);
   }
 
   std::optional<unsigned> weight(const Value *V) const {



More information about the llvm-commits mailing list