[llvm] [Draft] Data flow based shrink wrapping (PR #191784)

Elizaveta Noskova via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 28 08:46:17 PDT 2026


https://github.com/enoskova-sc updated https://github.com/llvm/llvm-project/pull/191784

>From 8afbad9b3c7ab44c428e34b57130710c54eed9cd Mon Sep 17 00:00:00 2001
From: Elizaveta Noskova <elizaveta.noskova at syntacore.com>
Date: Mon, 28 Sep 2026 15:08:56 +0300
Subject: [PATCH 1/2] [draft] Add data-flow ShrinkWrapping pass producing
 multiple save/restore points

Introduce a new machine-function pass that performs data-flow based shrink-wrapping and can split callee-saved register spills/restores into multiple save/restore points.

Extend the generic infrastructure required by the pass:
- MachineFrameInfo gains explicit Prolog/Epilog blocks, per-point callee-saved register info, and findSpilledIn/findRestoredIn helpers.
- PrologEpilogInserter tracks separate Prolog/Epilog and Save/Restore blocks and reconstructs per-block callee-saved info for multiple points.
- New target hooks (TargetRegisterInfo::isCSIFrameIndex/getCSIFrameOffset, TargetFrameLowering::enableCSRSaveRestorePointsSplit) with default implementations.
- Relax single-save/restore-point assertions in the AMDGPU and PowerPC frame lowering code and llvm-reduce.

The pass is registered but not added to the default codegen pipeline, so it does not change code generation unless selected explicitly (e.g. via -run-pass).
---
 llvm/include/llvm/CodeGen/MachineFrameInfo.h  |   49 +-
 llvm/include/llvm/CodeGen/Passes.h            |    2 +
 llvm/include/llvm/CodeGen/ShrinkWrapping.h    |   29 +
 .../llvm/CodeGen/TargetFrameLowering.h        |    4 +
 .../include/llvm/CodeGen/TargetRegisterInfo.h |    6 +
 llvm/include/llvm/InitializePasses.h          |    1 +
 llvm/lib/CodeGen/CMakeLists.txt               |    1 +
 llvm/lib/CodeGen/CodeGen.cpp                  |    1 +
 llvm/lib/CodeGen/PrologEpilogInserter.cpp     |  345 +++-
 llvm/lib/CodeGen/ShrinkWrapping.cpp           | 1732 +++++++++++++++++
 llvm/lib/Passes/PassBuilder.cpp               |    1 +
 llvm/lib/Target/AMDGPU/SILowerSGPRSpills.cpp  |   27 +-
 llvm/lib/Target/PowerPC/PPCFrameLowering.cpp  |    2 -
 .../CodeGen/AArch64/live-debugvalues-sve.mir  |   12 +-
 llvm/tools/llvm-reduce/ReducerWorkItem.cpp    |    6 -
 15 files changed, 2086 insertions(+), 132 deletions(-)
 create mode 100644 llvm/include/llvm/CodeGen/ShrinkWrapping.h
 create mode 100644 llvm/lib/CodeGen/ShrinkWrapping.cpp

diff --git a/llvm/include/llvm/CodeGen/MachineFrameInfo.h b/llvm/include/llvm/CodeGen/MachineFrameInfo.h
index 30699c3c18f1c4..f6815a05b2c0f2 100644
--- a/llvm/include/llvm/CodeGen/MachineFrameInfo.h
+++ b/llvm/include/llvm/CodeGen/MachineFrameInfo.h
@@ -348,9 +348,16 @@ class MachineFrameInfo {
   /// stack objects like arguments so we can't treat them as immutable.
   bool HasTailCall = false;
 
-  /// Not empty, if shrink-wrapping found a better place for the prologue.
+  /// Not null, if shrink-wrapping found a better place for the prologue.
+  MachineBasicBlock *Prolog = nullptr;
+  /// Not null, if shrink-wrapping found a better place for the epilogue.
+  MachineBasicBlock *Epilog = nullptr;
+
+  /// Not empty, if shrink-wrapping found a better place for saving callee
+  /// saves.
   SaveRestorePoints SavePoints;
-  /// Not empty, if shrink-wrapping found a better place for the epilogue.
+  /// Not empty, if shrink-wrapping found a better place for restoring callee
+  /// saves.
   SaveRestorePoints RestorePoints;
 
   /// Size of the UnsafeStack Frame
@@ -862,6 +869,39 @@ class MachineFrameInfo {
   /// \copydoc getCalleeSavedInfo()
   std::vector<CalleeSavedInfo> &getCalleeSavedInfo() { return CSInfo; }
 
+  /// Returns callee saved info vector for provided save point in
+  /// the current function.
+  std::vector<CalleeSavedInfo> getSaveCSInfo(MachineBasicBlock *MBB) const {
+    return SavePoints.lookup(MBB);
+  }
+
+  /// Returns callee saved info vector for provided restore point
+  /// in the current function.
+  const std::vector<CalleeSavedInfo>
+  getRestoreCSInfo(MachineBasicBlock *MBB) const {
+    return RestorePoints.lookup(MBB);
+  }
+
+  MachineBasicBlock *findSpilledIn(const CalleeSavedInfo &Match) const {
+    for (auto [BB, CSIV] : SavePoints) {
+      for (auto &CSI : CSIV) {
+        if (CSI.getReg() == Match.getReg())
+          return BB;
+      }
+    }
+    return nullptr;
+  }
+
+  MachineBasicBlock *findRestoredIn(const CalleeSavedInfo &Match) const {
+    for (auto [BB, CSIV] : RestorePoints) {
+      for (auto &CSI : CSIV) {
+        if (CSI.getReg() == Match.getReg())
+          return BB;
+      }
+    }
+    return nullptr;
+  }
+
   /// Used by prolog/epilog inserter to set the function's callee saved
   /// information.
   void setCalleeSavedInfo(std::vector<CalleeSavedInfo> CSI) {
@@ -885,6 +925,11 @@ class MachineFrameInfo {
     RestorePoints = std::move(NewRestorePoints);
   }
 
+  MachineBasicBlock *getProlog() const { return Prolog; }
+  void setProlog(MachineBasicBlock *BB) { Prolog = BB; }
+  MachineBasicBlock *getEpilog() const { return Epilog; }
+  void setEpilog(MachineBasicBlock *BB) { Epilog = BB; }
+
   void clearSavePoints() { SavePoints.clear(); }
   void clearRestorePoints() { RestorePoints.clear(); }
 
diff --git a/llvm/include/llvm/CodeGen/Passes.h b/llvm/include/llvm/CodeGen/Passes.h
index 861ca5aeeb8526..8627f5a2f78b31 100644
--- a/llvm/include/llvm/CodeGen/Passes.h
+++ b/llvm/include/llvm/CodeGen/Passes.h
@@ -192,6 +192,8 @@ LLVM_ABI extern char &SpillPlacementID;
 // instruction and update the MachineFunctionInfo with that information.
 LLVM_ABI extern char &ShrinkWrapID;
 
+LLVM_ABI extern char &ShrinkWrappingID;
+
 /// LiveRangeShrink pass. Move instruction close to its definition to shrink
 /// the definition's live range.
 LLVM_ABI extern char &LiveRangeShrinkID;
diff --git a/llvm/include/llvm/CodeGen/ShrinkWrapping.h b/llvm/include/llvm/CodeGen/ShrinkWrapping.h
new file mode 100644
index 00000000000000..202748a768d381
--- /dev/null
+++ b/llvm/include/llvm/CodeGen/ShrinkWrapping.h
@@ -0,0 +1,29 @@
+//===- llvm/CodeGen/ShrinkWrapping.h --------------------------------*- C++
+//-*-===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#ifndef LLVM_CODEGEN_SHRINKWRAPPING_H
+#define LLVM_CODEGEN_SHRINKWRAPPING_H
+
+#include "llvm/CodeGen/MachinePassManager.h"
+
+namespace llvm {
+
+class ShrinkWrappingPass : public RequiredPassInfoMixin<ShrinkWrappingPass> {
+public:
+  PreservedAnalyses run(MachineFunction &MF,
+                        MachineFunctionAnalysisManager &MFAM);
+
+  MachineFunctionProperties getRequiredProperties() const {
+    return MachineFunctionProperties().setNoVRegs();
+  }
+};
+
+} // namespace llvm
+
+#endif // LLVM_CODEGEN_SHRINKWRAPPING_H
diff --git a/llvm/include/llvm/CodeGen/TargetFrameLowering.h b/llvm/include/llvm/CodeGen/TargetFrameLowering.h
index beee49bfee2244..ba8702e9859c76 100644
--- a/llvm/include/llvm/CodeGen/TargetFrameLowering.h
+++ b/llvm/include/llvm/CodeGen/TargetFrameLowering.h
@@ -195,6 +195,10 @@ class LLVM_ABI TargetFrameLowering {
     return false;
   }
 
+  /// enableCSRSaveRestorePointsSplit - Returns true if the target support
+  /// multiple save/restore points in shrink wrapping.
+  virtual bool enableCSRSaveRestorePointsSplit() const { return false; }
+
   /// Returns true if the stack slot holes in the fixed and callee-save stack
   /// area should be used when allocating other stack locations to reduce stack
   /// size.
diff --git a/llvm/include/llvm/CodeGen/TargetRegisterInfo.h b/llvm/include/llvm/CodeGen/TargetRegisterInfo.h
index 6b0e3e1289c85d..087f57c2d3ca74 100644
--- a/llvm/include/llvm/CodeGen/TargetRegisterInfo.h
+++ b/llvm/include/llvm/CodeGen/TargetRegisterInfo.h
@@ -993,6 +993,12 @@ class LLVM_ABI TargetRegisterInfo : public MCRegisterInfo {
     return false;
   }
 
+  virtual bool isCSIFrameIndex(MachineFunction *MF, int FrameIndex) const {
+    return false;
+  }
+
+  virtual int64_t getCSIFrameOffset(MachineFunction *MF) const { return 0; }
+
   /// Process frame indices in reverse block order. This changes the behavior of
   /// the RegScavenger passed to eliminateFrameIndex. If this is true targets
   /// should scavengeRegisterBackwards in eliminateFrameIndex. New targets
diff --git a/llvm/include/llvm/InitializePasses.h b/llvm/include/llvm/InitializePasses.h
index d3f7f506733367..d59eef73f0bdbc 100644
--- a/llvm/include/llvm/InitializePasses.h
+++ b/llvm/include/llvm/InitializePasses.h
@@ -303,6 +303,7 @@ LLVM_ABI void
 initializeSeparateConstOffsetFromGEPLegacyPassPass(PassRegistry &);
 LLVM_ABI void initializeShadowStackGCLoweringPass(PassRegistry &);
 LLVM_ABI void initializeShrinkWrapLegacyPass(PassRegistry &);
+LLVM_ABI void initializeShrinkWrappingLegacyPass(PassRegistry &);
 LLVM_ABI void initializeSinkingLegacyPassPass(PassRegistry &);
 LLVM_ABI void initializeSjLjEHPreparePass(PassRegistry &);
 LLVM_ABI void initializeSlotIndexesWrapperPassPass(PassRegistry &);
diff --git a/llvm/lib/CodeGen/CMakeLists.txt b/llvm/lib/CodeGen/CMakeLists.txt
index 99dfb4bb09df73..c74df6ff7b6a94 100644
--- a/llvm/lib/CodeGen/CMakeLists.txt
+++ b/llvm/lib/CodeGen/CMakeLists.txt
@@ -233,6 +233,7 @@ add_llvm_component_library(LLVMCodeGen
   SelectOptimize.cpp
   ShadowStackGCLowering.cpp
   ShrinkWrap.cpp
+  ShrinkWrapping.cpp
   SjLjEHPrepare.cpp
   SlotIndexes.cpp
   SpillPlacement.cpp
diff --git a/llvm/lib/CodeGen/CodeGen.cpp b/llvm/lib/CodeGen/CodeGen.cpp
index efa9bdc635c5f1..8fba4655ca9be9 100644
--- a/llvm/lib/CodeGen/CodeGen.cpp
+++ b/llvm/lib/CodeGen/CodeGen.cpp
@@ -146,6 +146,7 @@ void llvm::initializeCodeGen(PassRegistry &Registry) {
   initializeSelectOptimizePass(Registry);
   initializeShadowStackGCLoweringPass(Registry);
   initializeShrinkWrapLegacyPass(Registry);
+  initializeShrinkWrappingLegacyPass(Registry);
   initializeSjLjEHPreparePass(Registry);
   initializeSlotIndexesWrapperPassPass(Registry);
   initializeStackColoringLegacyPass(Registry);
diff --git a/llvm/lib/CodeGen/PrologEpilogInserter.cpp b/llvm/lib/CodeGen/PrologEpilogInserter.cpp
index 7654014f2a1cbc..2f6fce1a29f0ed 100644
--- a/llvm/lib/CodeGen/PrologEpilogInserter.cpp
+++ b/llvm/lib/CodeGen/PrologEpilogInserter.cpp
@@ -79,8 +79,12 @@ namespace {
 class PEIImpl {
   RegScavenger *RS = nullptr;
 
-  // Save and Restore blocks of the current function. Typically there is a
-  // single save block, unless Windows EH funclets are involved.
+  // Prolog and Epilog blocks of the current function. Typically there is a
+  // single Prolog block, unless Windows EH funclets are involved.
+  MBBVector PrologBlocks;
+  MBBVector EpilogBlocks;
+
+  // Save and Restore blocks of the current function.
   MBBVector SaveBlocks;
   MBBVector RestoreBlocks;
 
@@ -98,6 +102,7 @@ class PEIImpl {
 
   void calculateCallFrameInfo(MachineFunction &MF);
   void calculateSaveRestoreBlocks(MachineFunction &MF);
+  void calculatePrologEpilogBlocks(MachineFunction &MF);
   void spillCalleeSavedRegs(MachineFunction &MF);
 
   void calculateFrameObjectOffsets(MachineFunction &MF);
@@ -226,14 +231,17 @@ bool PEIImpl::run(MachineFunction &MF) {
   // information. Also eliminates call frame pseudo instructions.
   calculateCallFrameInfo(MF);
 
-  // Determine placement of CSR spill/restore code and prolog/epilog code:
+  // Determine placement of CSR spill/restore code:
   // place all spills in the entry block, all restores in return blocks.
   calculateSaveRestoreBlocks(MF);
 
+  // Determine placement of prolog/epilog code.
+  calculatePrologEpilogBlocks(MF);
+
   // Stash away DBG_VALUEs that should not be moved by insertion of prolog code.
   SavedDbgValuesMap EntryDbgValues;
-  for (MachineBasicBlock *SaveBlock : SaveBlocks)
-    stashEntryDbgValues(*SaveBlock, EntryDbgValues);
+  for (MachineBasicBlock *PrologBlock : PrologBlocks)
+    stashEntryDbgValues(*PrologBlock, EntryDbgValues);
 
   // Handle CSR spilling and restoring, for targets that need it.
   if (MF.getTarget().usesPhysRegsForValues())
@@ -343,6 +351,8 @@ bool PEIImpl::run(MachineFunction &MF) {
   delete RS;
   SaveBlocks.clear();
   RestoreBlocks.clear();
+  PrologBlocks.clear();
+  EpilogBlocks.clear();
   MFI.clearSavePoints();
   MFI.clearRestorePoints();
   return true;
@@ -408,6 +418,38 @@ void PEIImpl::calculateCallFrameInfo(MachineFunction &MF) {
   }
 }
 
+/// Compute two sets of blocks for placing prolog and epilog code respectively.
+void PEIImpl::calculatePrologEpilogBlocks(MachineFunction &MF) {
+  const MachineFrameInfo &MFI = MF.getFrameInfo();
+  MachineBasicBlock *Prolog = MFI.getProlog();
+  MachineBasicBlock *Epilog = MFI.getEpilog();
+
+  // The data-flow ShrinkWrapping pass sets Prolog/Epilog explicitly and keeps
+  // the stack allocation at the entry and return blocks, spilling/restoring
+  // registers at the save/restore points.
+  //
+  // The classic ShrinkWrap pass only records save/restore points and leaves
+  // Prolog/Epilog unset. In that case the prolog/epilog must be emitted at the
+  // save/restore points, otherwise shrink-wrapping is lost and the call frame
+  // state can become inconsistent.
+  if (!Prolog && !Epilog) {
+    if (!SaveBlocks.empty())
+      PrologBlocks = SaveBlocks;
+    if (!RestoreBlocks.empty())
+      EpilogBlocks = RestoreBlocks;
+    return;
+  }
+
+  // Save refs to entry and return blocks.
+  PrologBlocks.push_back(&MF.front());
+  for (MachineBasicBlock &MBB : MF) {
+    if (MBB.isEHFuncletEntry())
+      PrologBlocks.push_back(&MBB);
+    if (MBB.isReturnBlock())
+      EpilogBlocks.push_back(&MBB);
+  }
+}
+
 /// Compute the sets of entry and return blocks for saving and restoring
 /// callee-saved registers, and placing prolog and epilog code.
 void PEIImpl::calculateSaveRestoreBlocks(MachineFunction &MF) {
@@ -418,19 +460,19 @@ void PEIImpl::calculateSaveRestoreBlocks(MachineFunction &MF) {
 
   // Use the points found by shrink-wrapping, if any.
   if (!MFI.getSavePoints().empty()) {
-    assert(MFI.getSavePoints().size() == 1 &&
-           "Multiple save points are not yet supported!");
-    const auto &SavePoint = *MFI.getSavePoints().begin();
-    SaveBlocks.push_back(SavePoint.first);
-    assert(MFI.getRestorePoints().size() == 1 &&
-           "Multiple restore points are not yet supported!");
-    const auto &RestorePoint = *MFI.getRestorePoints().begin();
-    MachineBasicBlock *RestoreBlock = RestorePoint.first;
-    // If RestoreBlock does not have any successor and is not a return block
-    // then the end point is unreachable and we do not need to insert any
-    // epilogue.
-    if (!RestoreBlock->succ_empty() || RestoreBlock->isReturnBlock())
-      RestoreBlocks.push_back(RestoreBlock);
+    assert(!MFI.getRestorePoints().empty() &&
+           "Both restore and save must be set");
+    for (auto &item : MFI.getSavePoints())
+      SaveBlocks.push_back(item.first);
+
+    for (auto &item : MFI.getRestorePoints()) {
+      MachineBasicBlock *RestoreBlock = item.first;
+      // If RestoreBlock does not have any successor and is not a return block
+      // then the end point is unreachable and we do not need to insert any
+      // epilogue.
+      if (!RestoreBlock->succ_empty() || RestoreBlock->isReturnBlock())
+        RestoreBlocks.push_back(RestoreBlock);
+    }
     return;
   }
 
@@ -536,8 +578,7 @@ static void assignCalleeSavedSpillSlots(MachineFunction &F,
 
 /// Helper function to update the liveness information for the callee-saved
 /// registers.
-static void updateLiveness(MachineFunction &MF) {
-  MachineFrameInfo &MFI = MF.getFrameInfo();
+static void updateLiveness(MachineFunction &MF, CalleeSavedInfo &Info) {
   // Visited will contain all the basic blocks that are in the region
   // where the callee saved registers are alive:
   // - Anything that is not Save or Restore -> LiveThrough.
@@ -545,73 +586,58 @@ static void updateLiveness(MachineFunction &MF) {
   // - Restore -> LiveOut.
   // The live-out is not attached to the block, so no need to keep
   // Restore in this set.
+  bool AddSucc = true;
   SmallPtrSet<MachineBasicBlock *, 8> Visited;
   SmallVector<MachineBasicBlock *, 8> WorkList;
   MachineBasicBlock *Entry = &MF.front();
+  MachineFrameInfo &MFI = MF.getFrameInfo();
 
-  assert(MFI.getSavePoints().size() < 2 &&
-         "Multiple save points not yet supported!");
-  MachineBasicBlock *Save = MFI.getSavePoints().empty()
-                                ? nullptr
-                                : (*MFI.getSavePoints().begin()).first;
-
-  if (!Save)
-    Save = Entry;
-
-  if (Entry != Save) {
-    WorkList.push_back(Entry);
-    Visited.insert(Entry);
-  }
-  Visited.insert(Save);
-
-  assert(MFI.getRestorePoints().size() < 2 &&
-         "Multiple restore points not yet supported!");
-  MachineBasicBlock *Restore = MFI.getRestorePoints().empty()
-                                   ? nullptr
-                                   : (*MFI.getRestorePoints().begin()).first;
-  if (Restore)
-    // By construction Restore cannot be visited, otherwise it
-    // means there exists a path to Restore that does not go
-    // through Save.
-    WorkList.push_back(Restore);
-
+  WorkList.push_back(Entry);
+  Visited.insert(Entry);
   while (!WorkList.empty()) {
+    AddSucc = true;
     const MachineBasicBlock *CurBB = WorkList.pop_back_val();
-    // By construction, the region that is after the save point is
-    // dominated by the Save and post-dominated by the Restore.
-    if (CurBB == Save && Save != Restore)
-      continue;
-    // Enqueue all the successors not already visited.
-    // Those are by construction either before Save or after Restore.
-    for (MachineBasicBlock *SuccBB : CurBB->successors())
-      if (Visited.insert(SuccBB).second)
-        WorkList.push_back(SuccBB);
-  }
+    auto SaveIt = MFI.getSavePoints().find(CurBB);
+    auto RestoreIt = MFI.getRestorePoints().find(CurBB);
+    if (SaveIt != MFI.getSavePoints().end() &&
+        count_if(SaveIt->second, [&Info](CalleeSavedInfo Other) {
+          return Other.getReg() == Info.getReg();
+        })) {
+      AddSucc = false;
+      if (RestoreIt != MFI.getRestorePoints().end() &&
+          count_if(RestoreIt->second, [&Info](CalleeSavedInfo Other) {
+            return Other.getReg() == Info.getReg();
+          }))
+        AddSucc = true;
+    }
 
-  const std::vector<CalleeSavedInfo> &CSI = MFI.getCalleeSavedInfo();
+    if (AddSucc) {
+      for (MachineBasicBlock *SuccBB : CurBB->successors())
+        if (Visited.insert(SuccBB).second)
+          WorkList.push_back(SuccBB);
+    }
+  }
 
   MachineRegisterInfo &MRI = MF.getRegInfo();
-  for (const CalleeSavedInfo &I : CSI) {
-    for (MachineBasicBlock *MBB : Visited) {
-      MCRegister Reg = I.getReg();
-      // Add the callee-saved register as live-in.
-      // It's killed at the spill.
-      if (!MRI.isReserved(Reg) && !MBB->isLiveIn(Reg))
-        MBB->addLiveIn(Reg);
-    }
-    // If callee-saved register is spilled to another register rather than
-    // spilling to stack, the destination register has to be marked as live for
-    // each MBB between the prologue and epilogue so that it is not clobbered
-    // before it is reloaded in the epilogue. The Visited set contains all
-    // blocks outside of the region delimited by prologue/epilogue.
-    if (I.isSpilledToReg()) {
-      for (MachineBasicBlock &MBB : MF) {
-        if (Visited.count(&MBB))
-          continue;
-        MCRegister DstReg = I.getDstReg();
-        if (!MBB.isLiveIn(DstReg))
-          MBB.addLiveIn(DstReg);
-      }
+  for (MachineBasicBlock *MBB : Visited) {
+    MCPhysReg Reg = Info.getReg();
+    // Add the callee-saved register as live-in.
+    // It's killed at the spill.
+    if (!MRI.isReserved(Reg) && !MBB->isLiveIn(Reg))
+      MBB->addLiveIn(Reg);
+  }
+  // If callee-saved register is spilled to another register rather than
+  // spilling to stack, the destination register has to be marked as live for
+  // each MBB between the save and restore point so that it is not clobbered
+  // before it is reloaded in the restore point. The Visited set contains all
+  // blocks outside of the region delimited by save/restore.
+  if (Info.isSpilledToReg()) {
+    for (MachineBasicBlock &MBB : MF) {
+      if (Visited.count(&MBB))
+        continue;
+      MCPhysReg DstReg = Info.getDstReg();
+      if (!MBB.isLiveIn(DstReg))
+        MBB.addLiveIn(DstReg);
     }
   }
 }
@@ -634,7 +660,7 @@ static void insertCSRSaves(MachineBasicBlock &SaveBlock,
 
 /// Insert restore code for the callee-saved registers used in the function.
 static void insertCSRRestores(MachineBasicBlock &RestoreBlock,
-                              std::vector<CalleeSavedInfo> &CSI) {
+                              std::vector<CalleeSavedInfo> CSI) {
   MachineFunction &MF = *RestoreBlock.getParent();
   const TargetInstrInfo *TII = MF.getSubtarget().getInstrInfo();
   const TargetFrameLowering *TFI = MF.getSubtarget().getFrameLowering();
@@ -651,6 +677,64 @@ static void insertCSRRestores(MachineBasicBlock &RestoreBlock,
   }
 }
 
+static void fillCSInfoPerBB(MachineFrameInfo &MFI,
+                            DenseMap<MCRegister, CalleeSavedInfo *> &RegToInfo,
+                            MBBVector &PrologEpilogBlocks, bool isSave) {
+  // Global CalleeSavedInfo list aggregating CSIVs for all points
+  std::vector<CalleeSavedInfo> GCSIV;
+  const SaveRestorePoints &SRPoints =
+      isSave ? MFI.getSavePoints() : MFI.getRestorePoints();
+  SaveRestorePoints Inner;
+  for (auto [BB, Regs] : SRPoints) {
+    // CalleeSavedInfo list for each point
+    std::vector<CalleeSavedInfo> CSIV;
+    for (auto &Reg : Regs) {
+      auto It = RegToInfo.find(Reg.getReg());
+      if (It == RegToInfo.end())
+        continue;
+      CSIV.push_back(*RegToInfo.at(Reg.getReg()));
+      GCSIV.push_back(*RegToInfo.at(Reg.getReg()));
+    }
+    // We need to sort CSIV, because Aarch64 expect CSI list to come sorted by
+    // frame index
+    sort(CSIV, [](const CalleeSavedInfo &Lhs, const CalleeSavedInfo &Rhs) {
+      return Lhs.getFrameIdx() < Rhs.getFrameIdx();
+    });
+    Inner.try_emplace(BB, std::move(CSIV));
+  }
+
+  // If in any case not all CSRs listed in MFI.getCalleeSavedInfo are in the
+  // list of spilled/restored registers (for example AArch64 backend add VG
+  // registers in the list of CalleeSavedRegs during spill slot assignment), we
+  // should add them to this list and spill/restore them in Prolog/Epilog.
+  if (GCSIV.size() < RegToInfo.size()) {
+    for (auto &RTI : RegToInfo) {
+      if (count_if(GCSIV, [&RTI](const CalleeSavedInfo &CSI) {
+            return CSI.getReg() == RTI.first;
+          }))
+        continue;
+      for (MachineBasicBlock *BB : PrologEpilogBlocks) {
+        if (auto Entry = Inner.find(BB); Entry != Inner.end()) {
+          auto &CSI = Entry->second;
+          CSI.push_back(*RTI.second);
+          sort(CSI, [](const CalleeSavedInfo &Lhs, const CalleeSavedInfo &Rhs) {
+            return Lhs.getFrameIdx() < Rhs.getFrameIdx();
+          });
+          continue;
+        }
+        // CalleeSavedInfo list for each point
+        Inner.try_emplace(BB,
+                          std::initializer_list<CalleeSavedInfo>{*RTI.second});
+      }
+    }
+  }
+
+  if (isSave)
+    MFI.setSavePoints(Inner);
+  else
+    MFI.setRestorePoints(Inner);
+}
+
 void PEIImpl::spillCalleeSavedRegs(MachineFunction &MF) {
   // We can't list this requirement in getRequiredProperties because some
   // targets (WebAssembly) use virtual registers past this point, and the pass
@@ -675,18 +759,24 @@ void PEIImpl::spillCalleeSavedRegs(MachineFunction &MF) {
     MFI.setCalleeSavedInfoValid(true);
 
     std::vector<CalleeSavedInfo> &CSI = MFI.getCalleeSavedInfo();
-
-    // Fill SavePoints and RestorePoints with CalleeSavedRegisters
-    if (!MFI.getSavePoints().empty()) {
-      SaveRestorePoints SaveRestorePts;
-      for (const auto &SavePoint : MFI.getSavePoints())
-        SaveRestorePts.insert({SavePoint.first, CSI});
-      MFI.setSavePoints(std::move(SaveRestorePts));
-
-      SaveRestorePts.clear();
-      for (const auto &RestorePoint : MFI.getRestorePoints())
-        SaveRestorePts.insert({RestorePoint.first, CSI});
-      MFI.setRestorePoints(std::move(SaveRestorePts));
+    DenseMap<MCRegister, CalleeSavedInfo *> RegToInfo;
+    for (auto &CS : CSI)
+      RegToInfo.insert({CS.getReg(), &CS});
+
+    // The classic ShrinkWrap pass records one save/restore point per block
+    // with an empty register list, while the data-flow ShrinkWrapping pass
+    // records the registers per point. Only the latter (or the multi-point
+    // case) needs the per-block CSI to be reconstructed.
+    bool UsePerBlockCSInfo = MFI.getSavePoints().size() > 1 ||
+                             MFI.getProlog() || MFI.getEpilog();
+    if (UsePerBlockCSInfo) {
+      fillCSInfoPerBB(MFI, RegToInfo, PrologBlocks, /*isSave=*/true);
+      fillCSInfoPerBB(MFI, RegToInfo, EpilogBlocks, /*isSave=*/false);
+    } else {
+      SaveRestorePoints SavePts;
+      for (MachineBasicBlock *PrologBlock : PrologBlocks)
+        SavePts.insert({PrologBlock, MFI.getCalleeSavedInfo()});
+      MFI.setSavePoints(std::move(SavePts));
     }
 
     if (!CSI.empty()) {
@@ -694,13 +784,32 @@ void PEIImpl::spillCalleeSavedRegs(MachineFunction &MF) {
         NumLeafFuncWithSpills++;
 
       for (MachineBasicBlock *SaveBlock : SaveBlocks)
-        insertCSRSaves(*SaveBlock, CSI);
+        insertCSRSaves(*SaveBlock, MFI.getSavePoints().empty()
+                                       ? CSI
+                                       : MFI.getSaveCSInfo(SaveBlock));
+
+      for (auto &CS : CSI) {
+        // Update the live-in information of all the blocks up to the save
+        // point.
+        updateLiveness(MF, CS);
+      }
 
-      // Update the live-in information of all the blocks up to the save point.
-      updateLiveness(MF);
+      if (!UsePerBlockCSInfo) {
+        SaveRestorePoints RestorePts;
+        for (MachineBasicBlock *EpilogBlock : EpilogBlocks)
+          RestorePts.insert({EpilogBlock, MFI.getCalleeSavedInfo()});
+        MFI.setRestorePoints(std::move(RestorePts));
+      }
 
-      for (MachineBasicBlock *RestoreBlock : RestoreBlocks)
-        insertCSRRestores(*RestoreBlock, CSI);
+      // Iterate over the restore blocks rather than over all blocks: when no
+      // restore points were recorded, getRestorePoints() is empty and the full
+      // CSI must be restored only at the actual restore blocks, not everywhere.
+      for (MachineBasicBlock *RestoreBlock : RestoreBlocks) {
+        insertCSRRestores(*RestoreBlock,
+                          MFI.getRestorePoints().empty()
+                              ? CSI
+                              : MFI.getRestoreCSInfo(RestoreBlock));
+      }
     }
   }
 }
@@ -1168,23 +1277,27 @@ void PEIImpl::insertPrologEpilogCode(MachineFunction &MF) {
   const TargetFrameLowering &TFI = *MF.getSubtarget().getFrameLowering();
 
   // Add prologue to the function...
-  for (MachineBasicBlock *SaveBlock : SaveBlocks)
-    TFI.emitPrologue(MF, *SaveBlock);
+  for (MachineBasicBlock *PrologBlock : PrologBlocks) {
+    TFI.emitPrologue(MF, *PrologBlock);
+  }
 
   // Add epilogue to restore the callee-save registers in each exiting block.
-  for (MachineBasicBlock *RestoreBlock : RestoreBlocks)
-    TFI.emitEpilogue(MF, *RestoreBlock);
+  for (MachineBasicBlock *EpilogBlock : EpilogBlocks) {
+    TFI.emitEpilogue(MF, *EpilogBlock);
+  }
 
-  for (MachineBasicBlock *SaveBlock : SaveBlocks)
-    TFI.inlineStackProbe(MF, *SaveBlock);
+  // insertZeroCallUsedRegs() runs later, from PEIImpl::run() after
+  // scavengeFrameVirtualRegs().
+  for (MachineBasicBlock *PrologBlock : PrologBlocks)
+    TFI.inlineStackProbe(MF, *PrologBlock);
 
   // Emit additional code that is required to support segmented stacks, if
   // we've been asked for it.  This, when linked with a runtime with support
   // for segmented stacks (libgcc is one), will result in allocating stack
   // space in small chunks instead of one large contiguous block.
   if (MF.shouldSplitStack()) {
-    for (MachineBasicBlock *SaveBlock : SaveBlocks)
-      TFI.adjustForSegmentedStacks(MF, *SaveBlock);
+    for (MachineBasicBlock *PrologBlock : PrologBlocks)
+      TFI.adjustForSegmentedStacks(MF, *PrologBlock);
   }
 
   // Emit additional code that is required to explicitly handle the stack in
@@ -1193,8 +1306,8 @@ void PEIImpl::insertPrologEpilogCode(MachineFunction &MF) {
   // different conditional check and another BIF for allocating more stack
   // space.
   if (MF.getFunction().getCallingConv() == CallingConv::HiPE)
-    for (MachineBasicBlock *SaveBlock : SaveBlocks)
-      TFI.adjustForHiPEPrologue(MF, *SaveBlock);
+    for (MachineBasicBlock *PrologBlock : PrologBlocks)
+      TFI.adjustForHiPEPrologue(MF, *PrologBlock);
 }
 
 /// insertZeroCallUsedRegs - Zero out call used registers.
@@ -1505,8 +1618,26 @@ void PEIImpl::replaceFrameIndicesBackward(MachineBasicBlock *BB,
       if (replaceFrameIndexDebugInstr(MF, MI, Idx, SPAdj))
         continue;
 
+      int FrameIndex = MI.getOperand(Idx).getIndex();
+      int SPA = 0;
+      // The extra callee-saved frame offset only applies when the data-flow
+      // ShrinkWrapping pass splits the save/restore points away from the
+      // prologue/epilogue. For the classic ShrinkWrap pass Prolog/Epilog are
+      // unset and the offset must not be applied; additionally the block sets
+      // may be stale here because inline stack probing can split blocks after
+      // calculatePrologEpilogBlocks().
+      MachineFrameInfo &MFI = MF.getFrameInfo();
+      if (TRI.isCSIFrameIndex(&MF, FrameIndex) &&
+          (MFI.getProlog() || MFI.getEpilog() ||
+           MFI.getSavePoints().size() > 1) &&
+          !(count(PrologBlocks, BB) || count(EpilogBlocks, BB))) {
+        SPA = TRI.getCSIFrameOffset(&MF);
+      }
+
+      if (SPA == 0)
+        SPA = SPAdj;
       // Eliminate this FrameIndex operand.
-      RemovedMI = TRI.eliminateFrameIndex(MI, SPAdj, Idx, LocalRS);
+      RemovedMI = TRI.eliminateFrameIndex(MI, SPA, Idx, LocalRS);
       if (RemovedMI)
         break;
     }
diff --git a/llvm/lib/CodeGen/ShrinkWrapping.cpp b/llvm/lib/CodeGen/ShrinkWrapping.cpp
new file mode 100644
index 00000000000000..6af6e6e5f409f3
--- /dev/null
+++ b/llvm/lib/CodeGen/ShrinkWrapping.cpp
@@ -0,0 +1,1732 @@
+//===-- ShrinkWrapping.cpp - Reduce spills/restores of callee-saved regs --===//
+//
+//                     The LLVM Compiler Infrastructure
+//
+// This file is distributed under the University of Illinois Open Source
+// License. See LICENSE.TXT for details.
+//
+//===----------------------------------------------------------------------===//
+//
+// This file implements a shrink wrapping variant of prolog/epilog insertion:
+// - Spills and restores of callee-saved registers (CSRs) are placed in the
+//   machine CFG to tightly surround their uses so that execution paths that
+//   do not use CSRs do not pay the spill/restore penalty.
+//
+// - Avoiding placment of spills/restores in loops: if a CSR is used inside a
+//   loop the spills are placed in the loop preheader, and restores are
+//   placed in the loop exit nodes (the successors of loop _exiting_ nodes).
+//
+// - Covering paths without CSR uses:
+//   If a region in a CFG uses CSRs and has multiple entry and/or exit points,
+//   the use info for the CSRs inside the region is propagated outward in the
+//   CFG to ensure validity of the spill/restore placements. This decreases
+//   the effectiveness of shrink wrapping but does not require edge splitting
+//   in the machine CFG.
+//
+// This shrink wrapping implementation uses an iterative analysis to determine
+// which basic blocks require spills and restores for CSRs.
+//
+// This pass uses MachineDominators and MachineLoopInfo. Loop information
+// is used to prevent placement of callee-saved register spills/restores
+// in the bodies of loops.
+//
+//===----------------------------------------------------------------------===//
+
+#define DEBUG_TYPE "shrink-wrapping"
+
+#include "llvm/CodeGen/ShrinkWrapping.h"
+#include "llvm/ADT/DenseMap.h"
+#include "llvm/ADT/DepthFirstIterator.h"
+#include "llvm/ADT/GraphTraits.h"
+#include "llvm/ADT/PostOrderIterator.h"
+#include "llvm/ADT/STLExtras.h"
+#include "llvm/ADT/SetVector.h"
+#include "llvm/ADT/SmallVector.h"
+#include "llvm/ADT/SparseBitVector.h"
+#include "llvm/ADT/Statistic.h"
+#include "llvm/Analysis/CFG.h"
+#include "llvm/Analysis/ValueTracking.h"
+#include "llvm/CodeGen/MachineBlockFrequencyInfo.h"
+#include "llvm/CodeGen/MachineDominators.h"
+#include "llvm/CodeGen/MachineFrameInfo.h"
+#include "llvm/CodeGen/MachineInstr.h"
+#include "llvm/CodeGen/MachineJumpTableInfo.h"
+#include "llvm/CodeGen/MachineLoopInfo.h"
+#include "llvm/CodeGen/MachinePostDominators.h"
+#include "llvm/CodeGen/MachineRegisterInfo.h"
+#include "llvm/CodeGen/RegisterClassInfo.h"
+#include "llvm/CodeGen/RegisterScavenging.h"
+#include "llvm/CodeGen/TargetFrameLowering.h"
+#include "llvm/CodeGen/TargetInstrInfo.h"
+#include "llvm/CodeGen/TargetLowering.h"
+#include "llvm/CodeGen/TargetRegisterInfo.h"
+#include "llvm/CodeGen/TargetSubtargetInfo.h"
+#include "llvm/InitializePasses.h"
+#include "llvm/MC/MCAsmInfo.h"
+#include "llvm/Support/CommandLine.h"
+#include "llvm/Support/Compiler.h"
+#include "llvm/Support/Debug.h"
+#include "llvm/Target/TargetMachine.h"
+#include <sstream>
+
+using namespace llvm;
+
+STATISTIC(numSRReduced, "Number of CSR spills+restores reduced.");
+STATISTIC(NumFunc, "Number of functions");
+STATISTIC(NumNotSaveOrRestore,
+          "Number of cases, in which one of the sets are empty.");
+STATISTIC(NumCandidates, "Number of shrink-wrapping candidates");
+STATISTIC(
+    NumFuncWithSplitting,
+    "Number of functions, for which we managed to split Save/Restore points");
+
+static cl::opt<cl::boolOrDefault>
+    EnableShrinkWrappingOpt("enable-shrink-wrapping", cl::Hidden,
+                            cl::desc("enable the shrink-wrapping pass"));
+static cl::opt<bool> EnableShrinkWrappingSplitOpt(
+    "enable-shrink-wrapping-into-multiple-points", cl::init(false), cl::Hidden,
+    cl::desc("enable splitting of the save and restore blocks if possible"));
+
+// Debugging level for shrink wrapping.
+enum ShrinkWrappingDebugLevel { Disabled, BasicInfo, Iterations, Details };
+
+static cl::opt<enum ShrinkWrappingDebugLevel> ShrinkWrappingDebugging(
+    "shrink-wrapping-dbg", cl::Hidden,
+    cl::desc("Print shrink wrapping debugging information"),
+    cl::values(clEnumValN(Disabled, "disable", "disable debug output"),
+               clEnumValN(BasicInfo, "basic", "print basic DF sets"),
+               clEnumValN(Iterations, "iters",
+                          "print SR sets for each iteration"),
+               clEnumValN(Details, "details", "print all DF sets")));
+
+struct AuxGraphNode {
+  std::string Name;
+  MachineBasicBlock *MatchMBB;
+  std::vector<AuxGraphNode *> Successors;
+  std::vector<AuxGraphNode *> Predecessors;
+
+  AuxGraphNode(MachineBasicBlock *MBB) : MatchMBB(MBB) {}
+
+  void addSuccessor(AuxGraphNode *S) {
+    Successors.push_back(S);
+    S->Predecessors.push_back(this);
+  }
+
+  std::string getBBName(const MachineBasicBlock *MBB) {
+    if (!MBB)
+      return "";
+
+    if (MBB->getBasicBlock())
+      return MBB->getBasicBlock()->getName().str();
+
+    std::ostringstream name;
+    name << "_MBB_" << MBB->getNumber();
+    return name.str();
+  }
+
+  auto succ_size() { return Successors.size(); }
+
+  auto succ_begin() { return Successors.begin(); }
+
+  auto succ_end() { return Successors.end(); }
+
+  auto pred_size() { return Predecessors.size(); }
+
+  auto pred_begin() { return Predecessors.begin(); }
+
+  auto pred_end() { return Predecessors.end(); }
+
+  void setupName() {
+    if (MatchMBB) {
+      Name = getBBName(MatchMBB);
+    } else {
+      assert(succ_size() == 1 && "Auxillary node has more than one successor!");
+      assert(pred_size() == 1 &&
+             "Auxillary node has more than one predecessor!");
+      MachineBasicBlock *Succ = (*succ_begin())->MatchMBB;
+      MachineBasicBlock *Pred = (*pred_begin())->MatchMBB;
+      assert(Succ && "Auxillary node should have real successor!");
+      assert(Pred && "Auxillary node should have real predecessor!");
+      std::string SuccName = getBBName(Succ);
+      std::string PredName = getBBName(Pred);
+      Name = PredName + "->" + SuccName;
+    }
+  }
+};
+
+struct AuxGraph {
+  std::vector<AuxGraphNode *> Nodes;
+  AuxGraphNode *Entry = nullptr;
+
+  void addNode(AuxGraphNode *N) {
+    Nodes.push_back(N);
+    if (!Entry)
+      Entry = N;
+  }
+
+  AuxGraphNode *getNode(MachineBasicBlock *MBB) {
+    auto It =
+        find_if(Nodes, [&MBB](AuxGraphNode *N) { return N->MatchMBB == MBB; });
+    if (It != Nodes.end())
+      return *It;
+    return nullptr;
+  }
+};
+
+namespace llvm {
+template <> struct GraphTraits<AuxGraphNode *> {
+  using NodeRef = AuxGraphNode *;
+  using ChildIteratorType = typename std::vector<AuxGraphNode *>::iterator;
+
+  static NodeRef getEntryNode(NodeRef N) { return N; }
+
+  static ChildIteratorType child_begin(NodeRef N) {
+    return N->Successors.begin();
+  }
+
+  static ChildIteratorType child_end(NodeRef N) { return N->Successors.end(); }
+};
+
+// Reverse traversal traits
+template <> struct llvm::GraphTraits<llvm::Inverse<AuxGraphNode *>> {
+  using NodeRef = AuxGraphNode *;
+  using ChildIteratorType = typename std::vector<AuxGraphNode *>::iterator;
+
+  static NodeRef getEntryNode(llvm::Inverse<NodeRef> N) { return N.Graph; }
+
+  // NOTE: You need a way to get predecessors!
+  // Simple approach: store them explicitly or compute on demand
+  static ChildIteratorType child_begin(NodeRef N) {
+    // In real code, maintain a Predecessors vector or compute via graph scan
+    return N->Predecessors.begin(); // Assume Predecessors exists
+  }
+
+  static ChildIteratorType child_end(NodeRef N) {
+    return N->Predecessors.end();
+  }
+};
+
+template <>
+struct GraphTraits<AuxGraph *> : public llvm::GraphTraits<AuxGraphNode *> {
+  using NodeRef = AuxGraphNode *;
+  using nodes_iterator = typename std::vector<AuxGraphNode *>::iterator;
+
+  static NodeRef getEntryNode(AuxGraph *G) { return G->Entry; }
+
+  static nodes_iterator nodes_begin(AuxGraph *G) { return G->Nodes.begin(); }
+
+  static nodes_iterator nodes_end(AuxGraph *G) { return G->Nodes.end(); }
+
+  static unsigned size(AuxGraph *G) { return G->Nodes.size(); }
+};
+} // namespace llvm
+
+namespace {
+
+/// Class to determine where the safe point to insert the
+/// prologue and epilogue are.
+/// Unlike the paper from Fred C. Chow, PLDI'88, that introduces the
+/// shrink-wrapping term for prologue/epilogue placement, this pass
+/// does not rely on expensive data-flow analysis. Instead we use the
+/// dominance properties and loop information to decide which point
+/// are safe for such insertion.
+class ShrinkWrappingImpl {
+  /// Hold callee-saved information.
+  RegisterClassInfo RCI;
+  MachineDominatorTree *MDT = nullptr;
+  MachinePostDominatorTree *MPDT = nullptr;
+
+  /// Hash table, mapping register with its corresponding spill and restore
+  /// basic block.
+  //  DenseMap<Register, std::pair<MachineBasicBlock *, MachineBasicBlock *>>
+  //    SavedRegs;
+  typedef SparseBitVector<> CSRegSet;
+  typedef DenseMap<MachineBasicBlock *, CSRegSet> CSRegBlockMap;
+  typedef DenseMap<AuxGraphNode *, CSRegSet> CSRegNodeMap;
+  CSRegSet UsedCSRegs;
+  CSRegBlockMap CSRUsed;
+  CSRegNodeMap AnticIn, AnticOut;
+  CSRegNodeMap AvailIn, AvailOut;
+  CSRegNodeMap CSRSave;
+  CSRegNodeMap CSRRestore;
+
+  /// Current opcode for frame setup.
+  unsigned FrameSetupOpcode = ~0u;
+
+  /// Current opcode for frame destroy.
+  unsigned FrameDestroyOpcode = ~0u;
+
+  /// Stack pointer register, used by llvm.{savestack,restorestack}
+  Register SP;
+
+  class SaveRestorePoints {
+    llvm::SaveRestorePoints SRPoints;
+
+  public:
+    llvm::SaveRestorePoints &get() { return SRPoints; }
+
+    void set(llvm::SaveRestorePoints &Rhs) { SRPoints = std::move(Rhs); }
+
+    void clear() { SRPoints.clear(); }
+
+    bool areMultiple() const { return SRPoints.size() > 1; }
+
+    MachineBasicBlock *getFirst() {
+      return SRPoints.empty() ? nullptr : SRPoints.begin()->first;
+    }
+
+    void insert(const std::pair<MachineBasicBlock *,
+                                std::vector<CalleeSavedInfo>> &Point) {
+      SRPoints.insert(Point);
+    }
+
+    void insert(
+        std::pair<MachineBasicBlock *, std::vector<CalleeSavedInfo>> &&Point) {
+      SRPoints.insert(Point);
+    }
+
+    std::vector<MachineBasicBlock *> insertReg(
+        Register Reg, MachineBasicBlock *MBB,
+        std::optional<std::vector<MachineBasicBlock *>> SaveRestoreBlockList) {
+      assert(MBB && "MBB is nullptr");
+      if (SRPoints.contains(MBB)) {
+        SRPoints[MBB].push_back(CalleeSavedInfo(Reg));
+        if (SaveRestoreBlockList.has_value())
+          return SaveRestoreBlockList.value();
+        return std::vector<MachineBasicBlock *>();
+      }
+      std::vector CSInfos{CalleeSavedInfo(Reg)};
+      SRPoints.insert(std::make_pair(MBB, CSInfos));
+      if (SaveRestoreBlockList.has_value()) {
+        SaveRestoreBlockList->push_back(MBB);
+        return SaveRestoreBlockList.value();
+      }
+      return std::vector<MachineBasicBlock *>();
+    }
+
+    void print(raw_ostream &OS, const TargetRegisterInfo *TRI) const {
+      for (auto [BB, CSIV] : SRPoints) {
+        OS << printMBBReference(*BB) << ": ";
+        for (auto &CSI : CSIV) {
+          OS << printReg(CSI.getReg(), TRI) << " ";
+        }
+        OS << "\n";
+      }
+    }
+
+    void dump(const TargetRegisterInfo *TRI) const { print(dbgs(), TRI); }
+  };
+
+  /// Class, wrapping hash table contained safe points, found for register spill
+  /// mapped to the list of corresponding registers. Register spill will be
+  /// inserted before the first instruction in this basic block.
+  SaveRestorePoints SavePoints;
+
+  /// Class, wrapping hash table contained safe points, found for register
+  /// restore mapped to the list of corresponding registers. Register restore
+  /// will be inserted before the first terminator instruction in this basic
+  /// block.
+  SaveRestorePoints RestorePoints;
+
+  std::vector<MachineBasicBlock *> SaveBlocks;
+  std::vector<MachineBasicBlock *> RestoreBlocks;
+
+  MachineBasicBlock *Prolog = nullptr;
+  MachineBasicBlock *Epilog = nullptr;
+
+  // Entry and return blocks of the current function.
+  SmallVector<MachineBasicBlock *, 4> ReturnBlocks;
+
+  /// Hold the loop information. Used to determine if Save and Restore
+  /// are in the same loop.
+  MachineLoopInfo *MLI = nullptr;
+
+  // Emit remarks.
+  MachineOptimizationRemarkEmitter *ORE = nullptr;
+
+  /// Entry block.
+  MachineBasicBlock *Entry = nullptr;
+
+  bool HasFastExitPath = false;
+
+  using SetOfRegs = SmallSetVector<unsigned, 16>;
+
+  /// Registers that need to be saved for the current function.
+  mutable SparseBitVector<> CurrentCSRsBitVec;
+
+  AuxGraph AuxillaryCFG;
+
+  /// Current MachineFunction.
+  MachineFunction *MachineFunc = nullptr;
+
+  /// Is `true` for the block numbers where we assume possible stack accesses
+  /// or computation of stack-relative addresses on any CFG path including the
+  /// block itself. Is `false` for basic blocks where we can guarantee the
+  /// opposite. False positives won't lead to incorrect analysis results,
+  /// therefore this approach is fair.
+  BitVector StackAddressUsedBlockInfo;
+
+  bool useOrDefCSR(const MachineInstr &MI, RegScavenger *RS,
+                   CSRegSet *RegsToSave) const;
+
+  /// Check if \p MI uses or defines a frame index.
+  /// If this is the case, this means \p MI must happen
+  /// after Save and before Restore.
+  bool useOrDefFI(const MachineInstr &MI, RegScavenger *RS,
+                  bool StackAddressUsed) const;
+
+  void createAuxillaryCFG();
+
+  bool calculateSets(MachineFunction &MF,
+                     const ReversePostOrderTraversal<MachineBasicBlock *> &RPOT,
+                     RegScavenger *RS);
+
+  std::string getBasicBlockName(const MachineBasicBlock *MBB) {
+    if (!MBB)
+      return "";
+
+    if (MBB->getBasicBlock())
+      return MBB->getBasicBlock()->getName().str();
+
+    std::ostringstream name;
+    name << "_MBB_" << MBB->getNumber();
+    return name.str();
+  }
+
+  const CSRegSet &getCurrentCSRsBitVec(RegScavenger *RS) const {
+    if (CurrentCSRsBitVec.empty()) {
+      BitVector SavedRegs;
+      const TargetFrameLowering *TFI =
+          MachineFunc->getSubtarget().getFrameLowering();
+
+      TFI->determineCalleeSaves(*MachineFunc, SavedRegs, RS);
+      for (unsigned Bit : SavedRegs.set_bits()) {
+        CurrentCSRsBitVec.set(Bit);
+      }
+    }
+    return CurrentCSRsBitVec;
+  }
+
+  void propagateUsesAroundLoop(MachineBasicBlock *MBB, MachineLoop *LP);
+  void verifySpillRestorePlacement();
+  void dumpUsed(MachineBasicBlock *MBB);
+
+  void dumpSet(const CSRegSet &s);
+
+  void dumpAllUsed();
+
+  void dumpSets(AuxGraphNode *Node);
+
+  void dumpAllSets();
+
+  void dumpSRSets();
+
+  std::string stringifyCSRegSet(const CSRegSet &s);
+
+  MachineBasicBlock *splitEdge(MachineBasicBlock *Pred,
+                               MachineBasicBlock *Succ);
+
+  void setupCFG();
+
+  void setupSaveRestorePoints();
+
+  void findFastExitPath();
+
+  void clearAnticAvailSets(RegScavenger *RS);
+
+  void clearAllSets(RegScavenger *RS);
+
+  void dumpSets1(AuxGraphNode *MBB);
+
+  bool calcAnticInOut(AuxGraphNode *Node);
+
+  bool calcAvailInOut(AuxGraphNode *Node);
+
+  void calculateAnticAvail(MachineFunction &Fn, RegScavenger *RS);
+
+  bool calcSpillPlacements(AuxGraphNode *Node,
+                           SmallVectorImpl<AuxGraphNode *> &blks,
+                           CSRegNodeMap &prevSpills);
+
+  bool calcRestorePlacements(AuxGraphNode *Node,
+                             SmallVectorImpl<AuxGraphNode *> &blks,
+                             CSRegNodeMap &prevRestores);
+
+  void placeSpillsAndRestores(MachineFunction &Fn);
+
+  /// Initialize the pass for \p MF.
+  void init(MachineFunction &MF, RegScavenger *RS) {
+    const TargetSubtargetInfo &Subtarget = MF.getSubtarget();
+    const TargetInstrInfo &TII = *Subtarget.getInstrInfo();
+    FrameSetupOpcode = TII.getCallFrameSetupOpcode();
+    FrameDestroyOpcode = TII.getCallFrameDestroyOpcode();
+    SP = Subtarget.getTargetLowering()->getStackPointerRegisterToSaveRestore();
+    RCI.runOnMachineFunction(MF);
+    SavePoints.clear();
+    RestorePoints.clear();
+    Prolog = nullptr;
+    Epilog = nullptr;
+    SaveBlocks.clear();
+    RestoreBlocks.clear();
+    Entry = &MF.front();
+    CurrentCSRsBitVec.clear();
+    MachineFunc = &MF;
+    clearAllSets(RS);
+    HasFastExitPath = false;
+    ++NumFunc;
+  }
+
+public:
+  ShrinkWrappingImpl(MachineDominatorTree *MDT, MachinePostDominatorTree *MPDT,
+                     MachineBlockFrequencyInfo *MBFI, MachineLoopInfo *MLI,
+                     MachineOptimizationRemarkEmitter *ORE)
+      : MDT(MDT), MPDT(MPDT), MLI(MLI), ORE(ORE) {}
+
+  /// Check if shrink wrapping is enabled for this target and function.
+  static bool isShrinkWrappingEnabled(const MachineFunction &MF);
+
+  bool run(MachineFunction &MF);
+};
+
+class ShrinkWrappingLegacy : public MachineFunctionPass {
+public:
+  static char ID;
+
+  ShrinkWrappingLegacy() : MachineFunctionPass(ID) {
+    initializeShrinkWrappingLegacyPass(*PassRegistry::getPassRegistry());
+  }
+
+  void getAnalysisUsage(AnalysisUsage &AU) const override {
+    AU.setPreservesAll();
+    AU.addRequired<MachineBlockFrequencyInfoWrapperPass>();
+    AU.addRequired<MachineDominatorTreeWrapperPass>();
+    AU.addRequired<MachinePostDominatorTreeWrapperPass>();
+    AU.addRequired<MachineLoopInfoWrapperPass>();
+    AU.addRequired<MachineOptimizationRemarkEmitterPass>();
+    MachineFunctionPass::getAnalysisUsage(AU);
+  }
+
+  MachineFunctionProperties getRequiredProperties() const override {
+    return MachineFunctionProperties().setNoVRegs();
+  }
+
+  StringRef getPassName() const override { return "Shrink Wrapping analysis"; }
+
+  /// Perform the shrink-wrapping analysis and update
+  /// the MachineFrameInfo attached to \p MF with the results.
+  bool runOnMachineFunction(MachineFunction &MF) override;
+};
+
+} // end anonymous namespace
+
+char ShrinkWrappingLegacy::ID = 0;
+
+char &llvm::ShrinkWrappingID = ShrinkWrappingLegacy::ID;
+
+INITIALIZE_PASS_BEGIN(ShrinkWrappingLegacy, DEBUG_TYPE, "Shrink Wrapping Pass",
+                      false, false)
+INITIALIZE_PASS_DEPENDENCY(MachineBlockFrequencyInfoWrapperPass)
+INITIALIZE_PASS_DEPENDENCY(MachineDominatorTreeWrapperPass)
+INITIALIZE_PASS_DEPENDENCY(MachinePostDominatorTreeWrapperPass)
+INITIALIZE_PASS_DEPENDENCY(MachineLoopInfoWrapperPass)
+INITIALIZE_PASS_DEPENDENCY(MachineOptimizationRemarkEmitterPass)
+INITIALIZE_PASS_END(ShrinkWrappingLegacy, DEBUG_TYPE, "Shrink Wrapping Pass",
+                    false, false)
+
+static bool giveUpWithRemarks(MachineOptimizationRemarkEmitter *ORE,
+                              StringRef RemarkName, StringRef RemarkMessage,
+                              const DiagnosticLocation &Loc,
+                              const MachineBasicBlock *MBB) {
+  ORE->emit([&]() {
+    return MachineOptimizationRemarkMissed(DEBUG_TYPE, RemarkName, Loc, MBB)
+           << RemarkMessage;
+  });
+
+  LLVM_DEBUG(dbgs() << RemarkMessage << '\n');
+  return false;
+}
+
+void ShrinkWrappingImpl::createAuxillaryCFG() {
+  std::set<MachineBasicBlock *> Visited;
+  SmallVector<MachineBasicBlock *, 8> WorkList;
+  WorkList.push_back(Entry);
+  Visited.insert(Entry);
+  auto *EntryNode = new AuxGraphNode(Entry);
+  AuxillaryCFG.addNode(EntryNode);
+  while (!WorkList.empty()) {
+    MachineBasicBlock *MBB = WorkList.pop_back_val();
+    AuxGraphNode *Node = nullptr;
+    if (MBB == Entry) {
+      Node = EntryNode;
+    } else {
+      Node = AuxillaryCFG.getNode(MBB);
+    }
+    assert(Node && "Node is null!");
+    for (MachineBasicBlock *Succ : MBB->successors()) {
+      auto *AuxNode = new AuxGraphNode(nullptr);
+      Node->addSuccessor(AuxNode);
+      AuxGraphNode *SuccNode = nullptr;
+      if (!Visited.insert(Succ).second) {
+        SuccNode = AuxillaryCFG.getNode(Succ);
+      } else {
+        SuccNode = new AuxGraphNode(Succ);
+        AuxillaryCFG.addNode(SuccNode);
+        WorkList.push_back(Succ);
+      }
+      AuxNode->addSuccessor(SuccNode);
+      AuxillaryCFG.addNode(AuxNode);
+    }
+  }
+
+  for (auto &Node : AuxillaryCFG.Nodes)
+    Node->setupName();
+}
+
+/// findFastExitPath - debugging method used to detect functions
+/// with at least one path from the entry block to a return block
+/// directly or which has a very small number of edges.
+///
+void ShrinkWrappingImpl::findFastExitPath() {
+  if (!Entry)
+    return;
+  // Fina a path from EntryBlock to any return block that does not branch:
+  //        Entry
+  //          |     ...
+  //          v      |
+  //         B1<-----+
+  //          |
+  //          v
+  //       Return
+  for (MachineBasicBlock::succ_iterator SI = Entry->succ_begin(),
+                                        SE = Entry->succ_end();
+       SI != SE; ++SI) {
+    MachineBasicBlock *SUCC = *SI;
+
+    // Assume positive, disprove existence of fast path.
+    HasFastExitPath = true;
+
+    // Check the immediate successors.
+    if (SUCC->isReturnBlock()) {
+      if (ShrinkWrappingDebugging >= BasicInfo)
+        dbgs() << "Fast exit path: " << printMBBReference(*Entry) << "->"
+               << printMBBReference(*SUCC) << "\n";
+      break;
+    }
+    // Traverse df from SUCC, look for a branch block.
+    std::string exitPath = getBasicBlockName(SUCC);
+    for (df_iterator<MachineBasicBlock *> BI = df_begin(SUCC),
+                                          BE = df_end(SUCC);
+         BI != BE; ++BI) {
+      MachineBasicBlock *SBB = *BI;
+      // Reject paths with branch nodes.
+      if (SBB->succ_size() > 1) {
+        HasFastExitPath = false;
+        break;
+      }
+      exitPath += "->" + getBasicBlockName(SBB);
+    }
+    if (HasFastExitPath) {
+      if (ShrinkWrappingDebugging >= BasicInfo)
+        dbgs() << "Fast exit path: " << getBasicBlockName(Entry) << "->"
+               << exitPath << "\n";
+      break;
+    }
+  }
+}
+
+/// verifySpillRestorePlacement - check the current spill/restore
+/// sets for safety. Attempt to find spills without restores or
+/// restores without spills.
+/// Spills: walk df from each MBB in spill set ensuring that
+///         all CSRs spilled at MMBB are restored on all paths
+///         from MBB to all exit blocks.
+/// Restores: walk idf from each MBB in restore set ensuring that
+///           all CSRs restored at MBB are spilled on all paths
+///           reaching MBB.
+///
+void ShrinkWrappingImpl::verifySpillRestorePlacement() {
+  for (CSRegNodeMap::iterator BI = CSRSave.begin(), BE = CSRSave.end();
+       BI != BE; ++BI) {
+    MachineBasicBlock *MBB = BI->first->MatchMBB;
+    AuxGraphNode *Node = BI->first;
+    CSRegSet spilled = BI->second;
+    CSRegSet restored;
+
+    if (spilled.empty())
+      continue;
+
+    LLVM_DEBUG(dbgs() << "SAVE[" << getBasicBlockName(MBB)
+                      << "] = " << stringifyCSRegSet(spilled) << "  RESTORE["
+                      << getBasicBlockName(MBB)
+                      << "] = " << stringifyCSRegSet(CSRRestore[Node]) << "\n");
+
+    if (CSRRestore[Node].intersects(spilled)) {
+      restored |= (CSRRestore[Node] & spilled);
+    }
+
+    // Walk depth first from MBB to find restores of all CSRs spilled at MBB:
+    // we must find restores for all spills w/no intervening spills on all
+    // paths from MBB to all return blocks.
+    for (df_iterator<MachineBasicBlock *> BI = df_begin(MBB), BE = df_end(MBB);
+         BI != BE; ++BI) {
+      MachineBasicBlock *SBB = *BI;
+      if (SBB == MBB)
+        continue;
+      // Stop when we encounter spills of any CSRs spilled at MBB that
+      // have not yet been seen to be restored.
+      AuxGraphNode *SBBNode = AuxillaryCFG.getNode(SBB);
+      if (CSRSave[SBBNode].intersects(spilled) &&
+          !restored.contains(CSRSave[SBBNode] & spilled))
+        break;
+      // Collect the CSRs spilled at MBB that are restored
+      // at this DF successor of MBB.
+
+      if (CSRRestore[SBBNode].intersects(spilled))
+        restored |= (CSRRestore[SBBNode] & spilled);
+      // If we are at a retun block, check that the restores
+      // we have seen so far exhaust the spills at MBB, then
+      // reset the restores.
+      if (SBB->isReturnBlock()) {
+        if (restored != spilled) {
+          CSRegSet notRestored = (spilled - restored);
+          LLVM_DEBUG(dbgs() << MachineFunc->getName() << ": "
+                            << stringifyCSRegSet(notRestored) << " spilled at "
+                            << getBasicBlockName(MBB)
+                            << " are never restored on path to return "
+                            << getBasicBlockName(SBB) << "\n");
+        }
+        SparseBitVector<> SBBRestored = CSRRestore[SBBNode] & spilled;
+        for (unsigned Bit : SBBRestored)
+          restored.reset(Bit);
+      }
+    }
+  }
+
+  // Check restore placements.
+  for (CSRegNodeMap::iterator BI = CSRRestore.begin(), BE = CSRRestore.end();
+       BI != BE; ++BI) {
+    MachineBasicBlock *MBB = BI->first->MatchMBB;
+    AuxGraphNode *Node = BI->first;
+    CSRegSet restored = BI->second;
+    CSRegSet spilled;
+
+    if (restored.empty())
+      continue;
+
+    LLVM_DEBUG(dbgs() << "SAVE[" << getBasicBlockName(MBB)
+                      << "] = " << stringifyCSRegSet(CSRSave[Node])
+                      << "  RESTORE[" << getBasicBlockName(MBB)
+                      << "] = " << stringifyCSRegSet(restored) << "\n");
+
+    if (CSRSave[Node].intersects(restored)) {
+      spilled |= (CSRSave[Node] & restored);
+    }
+    // Walk inverse depth first from MBB to find spills of all
+    // CSRs restored at MBB:
+    for (idf_iterator<MachineBasicBlock *> BI = idf_begin(MBB),
+                                           BE = idf_end(MBB);
+         BI != BE; ++BI) {
+      MachineBasicBlock *PBB = *BI;
+      if (PBB == MBB)
+        continue;
+      AuxGraphNode *PBBNode = AuxillaryCFG.getNode(PBB);
+      // Stop when we encounter restores of any CSRs restored at MBB that
+      // have not yet been seen to be spilled.
+      if (CSRRestore[PBBNode].intersects(restored) &&
+          !spilled.contains(CSRRestore[PBBNode] & restored))
+        break;
+      // Collect the CSRs restored at MBB that are spilled
+      // at this DF predecessor of MBB.
+      if (CSRSave[PBBNode].intersects(restored))
+        spilled |= (CSRSave[PBBNode] & restored);
+    }
+    if (spilled != restored) {
+      CSRegSet notSpilled = (restored - spilled);
+      LLVM_DEBUG(dbgs() << MachineFunc->getName() << ": "
+                        << stringifyCSRegSet(notSpilled) << " restored at "
+                        << printMBBReference(*MBB) << " are never spilled\n");
+    }
+  }
+}
+
+std::string ShrinkWrappingImpl::stringifyCSRegSet(const CSRegSet &s) {
+  const TargetRegisterInfo *TRI = MachineFunc->getSubtarget().getRegisterInfo();
+  std::ostringstream srep;
+
+  if (s.empty()) {
+    srep << "[]";
+    return srep.str();
+  }
+
+  srep << "[";
+  for (unsigned Reg : s) {
+    srep << ",";
+    srep << TRI->getName(Reg);
+  }
+  srep << "]";
+  return srep.str();
+}
+
+void ShrinkWrappingImpl::dumpSet(const CSRegSet &s) {
+  LLVM_DEBUG(dbgs() << stringifyCSRegSet(s) << "\n");
+}
+
+void ShrinkWrappingImpl::dumpUsed(MachineBasicBlock *MBB) {
+  LLVM_DEBUG({
+    if (MBB)
+      dbgs() << "CSRUsed[" << getBasicBlockName(MBB)
+             << "] = " << stringifyCSRegSet(CSRUsed[MBB]) << "\n";
+  });
+}
+
+void ShrinkWrappingImpl::dumpAllUsed() {
+  for (MachineFunction::iterator MBBI = MachineFunc->begin(),
+                                 MBBE = MachineFunc->end();
+       MBBI != MBBE; ++MBBI) {
+    dumpUsed(&(*MBBI));
+  }
+}
+
+void ShrinkWrappingImpl::dumpSets(AuxGraphNode *Node) {
+  LLVM_DEBUG({
+    if (Node) {
+      CSRegSet Used;
+      if (Node->MatchMBB)
+        Used = CSRUsed[Node->MatchMBB];
+      dbgs() << Node->Name << " | " << stringifyCSRegSet(Used) << " | "
+             << stringifyCSRegSet(AnticIn[Node]) << " | "
+             << stringifyCSRegSet(AnticOut[Node]) << " | "
+             << stringifyCSRegSet(AvailIn[Node]) << " | "
+             << stringifyCSRegSet(AvailOut[Node]) << "\n";
+    }
+  });
+}
+
+void ShrinkWrappingImpl::dumpSets1(AuxGraphNode *Node) {
+  LLVM_DEBUG({
+    if (Node) {
+      CSRegSet Used;
+      if (Node->MatchMBB)
+        Used = CSRUsed[Node->MatchMBB];
+      dbgs() << Node->Name << " | " << stringifyCSRegSet(Used) << " | "
+             << stringifyCSRegSet(AnticIn[Node]) << " | "
+             << stringifyCSRegSet(AnticOut[Node]) << " | "
+             << stringifyCSRegSet(AvailIn[Node]) << " | "
+             << stringifyCSRegSet(AvailOut[Node]) << " | "
+             << stringifyCSRegSet(CSRSave[Node]) << " | "
+             << stringifyCSRegSet(CSRRestore[Node]) << "\n";
+    }
+  });
+}
+
+void ShrinkWrappingImpl::dumpAllSets() {
+  for (auto &Node : AuxillaryCFG.Nodes) {
+    dumpSets1(Node);
+  }
+}
+
+void ShrinkWrappingImpl::dumpSRSets() {
+  LLVM_DEBUG({
+    for (auto &Node : AuxillaryCFG.Nodes) {
+      if (!CSRSave[Node].empty()) {
+        dbgs() << "SAVE[" << Node->Name
+               << "] = " << stringifyCSRegSet(CSRSave[Node]);
+        dbgs() << '\n';
+      }
+
+      if (!CSRRestore[Node].empty())
+        dbgs() << "RESTORE[" << Node->Name
+               << "] = " << stringifyCSRegSet(CSRRestore[Node]) << "\n";
+    }
+  });
+}
+
+// Initialize shrink wrapping DFA sets, called before iterations.
+void ShrinkWrappingImpl::clearAnticAvailSets(RegScavenger *RS) {
+  for (auto &Node : AuxillaryCFG.Nodes) {
+    AnticIn[Node] = getCurrentCSRsBitVec(RS);
+    AvailOut[Node] = getCurrentCSRsBitVec(RS);
+  }
+  AnticOut.clear();
+  AvailIn.clear();
+}
+
+// Clear all sets constructed by shrink wrapping.
+void ShrinkWrappingImpl::clearAllSets(RegScavenger *RS) {
+  ReturnBlocks.clear();
+  clearAnticAvailSets(RS);
+  UsedCSRegs.clear();
+  CSRUsed.clear();
+  CSRSave.clear();
+  CSRRestore.clear();
+}
+
+/// propagateUsesAroundLoop - copy used register info from MBB to all blocks
+/// of the loop given by LP and its parent loops. This prevents spills/restores
+/// from being placed in the bodies of loops.
+///
+void ShrinkWrappingImpl::propagateUsesAroundLoop(MachineBasicBlock *MBB,
+                                                 MachineLoop *LP) {
+  if (!MBB || !LP)
+    return;
+
+  std::vector<MachineBasicBlock *> loopBlocks = LP->getBlocks();
+  for (unsigned i = 0, e = loopBlocks.size(); i != e; ++i) {
+    MachineBasicBlock *LBB = loopBlocks[i];
+    if (LBB == MBB)
+      continue;
+    if (CSRUsed[LBB].contains(CSRUsed[MBB]))
+      continue;
+    CSRUsed[LBB] |= CSRUsed[MBB];
+  }
+}
+
+/// calcAnticInOut - calculate the anticipated in/out reg sets
+/// for the given MBB by looking forward in the MCFG at MBB's
+/// successors.
+///
+bool ShrinkWrappingImpl::calcAnticInOut(AuxGraphNode *Node) {
+  bool changed = false;
+
+  // AnticOut[MBB] = INTERSECT(AnticIn[S] for S in SUCCESSORS(MBB))
+  SmallVector<AuxGraphNode *, 4> successors;
+  for (auto SI = Node->succ_begin(), SE = Node->succ_end(); SI != SE; ++SI) {
+    AuxGraphNode *SUCC = *SI;
+    if (SUCC != Node)
+      successors.push_back(SUCC);
+  }
+
+  unsigned i = 0, e = successors.size();
+  if (i != e) {
+    CSRegSet prevAnticOut = AnticOut[Node];
+    AuxGraphNode *SUCC = successors[i];
+
+    AnticOut[Node] = AnticIn[SUCC];
+    for (++i; i != e; ++i) {
+      SUCC = successors[i];
+      AnticOut[Node] &= AnticIn[SUCC];
+    }
+    if (prevAnticOut != AnticOut[Node])
+      changed = true;
+  }
+
+  // AnticIn[MBB] = UNION(CSRUsed[MBB], AnticOut[MBB]);
+  CSRegSet prevAnticIn = AnticIn[Node];
+  CSRegSet UsedRegSet;
+  if (Node->MatchMBB)
+    UsedRegSet = CSRUsed[Node->MatchMBB];
+
+  AnticIn[Node] = UsedRegSet | AnticOut[Node];
+  if (prevAnticIn != AnticIn[Node])
+    changed = true;
+  return changed;
+}
+
+/// calcAvailInOut - calculate the available in/out reg sets
+/// for the given MBB by looking backward in the MCFG at MBB's
+/// predecessors.
+///
+bool ShrinkWrappingImpl::calcAvailInOut(AuxGraphNode *Node) {
+  bool changed = false;
+
+  // AvailIn[MBB] = INTERSECT(AvailOut[P] for P in PREDECESSORS(MBB))
+  SmallVector<AuxGraphNode *, 4> predecessors;
+  for (auto PI = Node->pred_begin(), PE = Node->pred_end(); PI != PE; ++PI) {
+    AuxGraphNode *PRED = *PI;
+    if (PRED != Node)
+      predecessors.push_back(PRED);
+  }
+
+  unsigned i = 0, e = predecessors.size();
+  if (i != e) {
+    CSRegSet prevAvailIn = AvailIn[Node];
+    AuxGraphNode *PRED = predecessors[i];
+
+    AvailIn[Node] = AvailOut[PRED];
+    for (++i; i != e; ++i) {
+      PRED = predecessors[i];
+      AvailIn[Node] &= AvailOut[PRED];
+    }
+    if (prevAvailIn != AvailIn[Node])
+      changed = true;
+  }
+
+  // AvailOut[MBB] = UNION(CSRUsed[MBB], AvailIn[MBB]);
+  CSRegSet prevAvailOut = AvailOut[Node];
+  CSRegSet UsedRegSet;
+  if (Node->MatchMBB)
+    UsedRegSet = CSRUsed[Node->MatchMBB];
+  AvailOut[Node] = UsedRegSet | AvailIn[Node];
+  if (prevAvailOut != AvailOut[Node])
+    changed = true;
+  return changed;
+}
+
+/// calculateAnticAvail - build the sets anticipated and available
+/// registers in the MCFG of the current function iteratively,
+/// doing a combined forward and backward analysis.
+///
+void ShrinkWrappingImpl::calculateAnticAvail(MachineFunction &Fn,
+                                             RegScavenger *RS) {
+  // Initialize data flow sets.
+  clearAnticAvailSets(RS);
+
+  // Calculate Antic{In,Out} and Avail{In,Out} iteratively on the MCFG.
+  bool changed = true;
+  unsigned iterations = 0;
+  while (changed) {
+    changed = false;
+    ++iterations;
+    for (auto &Node : AuxillaryCFG.Nodes) {
+      // Calculate anticipated in, out regs at MBB from
+      // anticipated at successors of MBB.
+      changed |= calcAnticInOut(Node);
+
+      // Calculate available in, out regs at MBB from
+      // available at predecessors of MBB.
+      changed |= calcAvailInOut(Node);
+    }
+  }
+
+  LLVM_DEBUG({
+    if (ShrinkWrappingDebugging >= Details) {
+      dbgs() << "-----------------------------------------------------------\n"
+             << " Antic/Avail Sets:\n"
+             << "-----------------------------------------------------------\n"
+             << "iterations = " << iterations << "\n"
+             << "-----------------------------------------------------------\n"
+             << "MBB | USED | ANTIC_IN | ANTIC_OUT | AVAIL_IN | AVAIL_OUT\n"
+             << "-----------------------------------------------------------\n";
+
+      for (auto &Node : AuxillaryCFG.Nodes)
+        dumpSets(Node);
+
+      dbgs() << "-----------------------------------------------------------\n";
+    }
+  });
+}
+
+bool ShrinkWrappingImpl::useOrDefCSR(const MachineInstr &MI, RegScavenger *RS,
+                                     CSRegSet *RegsToSave) const {
+  const MachineFunction *MF = MI.getParent()->getParent();
+  const TargetRegisterInfo *TRI = MF->getSubtarget().getRegisterInfo();
+  for (const MachineOperand &MO : MI.operands()) {
+    if (MO.isReg()) {
+      // Ignore instructions like DBG_VALUE which don't read/def the register.
+      if (!MO.isDef() && !MO.readsReg())
+        continue;
+      Register PhysReg = MO.getReg();
+      if (!PhysReg)
+        continue;
+      assert(PhysReg.isPhysical() && "Unallocated register?!");
+      // The stack pointer is not normally described as a callee-saved register
+      // in calling convention definitions, so we need to watch for it
+      // separately. An SP mentioned by a call instruction, we can ignore,
+      // though, as it's harmless and we do not want to effectively disable tail
+      // calls by forcing the restore point to post-dominate them.
+      // PPC's LR is also not normally described as a callee-saved register in
+      // calling convention definitions, so we need to watch for it, too. An LR
+      // mentioned implicitly by a return (or "branch to link register")
+      // instruction we can ignore, otherwise we may pessimize shrinkwrapping.
+      // PPC's Frame pointer (FP) is also not described as a callee-saved
+      // register. Until the FP is assigned a Physical Register PPC's FP needs
+      // to be checked separately.
+      if ((!MI.isCall() && PhysReg == SP) ||
+          RCI.getLastCalleeSavedAlias(PhysReg) ||
+          (!MI.isReturn() &&
+           TRI->isNonallocatableRegisterCalleeSave(PhysReg)) ||
+          TRI->isVirtualFrameRegister(PhysReg)) {
+        LLVM_DEBUG(dbgs() << MI << " uses or defines CSR: "
+                          << RCI.getLastCalleeSavedAlias(PhysReg) << "\n");
+        if (!RegsToSave)
+          return true;
+
+        RegsToSave->set(RCI.getLastCalleeSavedAlias(PhysReg));
+      }
+    } else if (MO.isRegMask()) {
+      // Check if this regmask clobbers any of the CSRs.
+      for (unsigned Reg : getCurrentCSRsBitVec(RS)) {
+        if (MO.clobbersPhysReg(Reg)) {
+          if (!RegsToSave)
+            return true;
+          RegsToSave->set(RCI.getLastCalleeSavedAlias(Reg));
+        }
+      }
+    }
+  }
+
+  // Skip FrameIndex operands in DBG_VALUE instructions.
+  if (RegsToSave && !RegsToSave->empty()) {
+    return true;
+  }
+  return false;
+}
+
+bool ShrinkWrappingImpl::useOrDefFI(const MachineInstr &MI, RegScavenger *RS,
+                                    bool StackAddressUsed) const {
+  /// Check if \p Op is known to access an address not on the function's stack .
+  /// At the moment, accesses where the underlying object is a global, function
+  /// argument, or jump table are considered non-stack accesses. Note that the
+  /// caller's stack may get accessed when passing an argument via the stack,
+  /// but not the stack of the current function.
+  ///
+  auto IsKnownNonStackPtr = [](MachineMemOperand *Op) {
+    if (Op->getValue()) {
+      const Value *UO = getUnderlyingObject(Op->getValue());
+      if (!UO)
+        return false;
+      if (auto *Arg = dyn_cast<Argument>(UO))
+        return !Arg->hasPassPointeeByValueCopyAttr();
+      return isa<GlobalValue>(UO);
+    }
+    if (const PseudoSourceValue *PSV = Op->getPseudoValue())
+      return PSV->isJumpTable() || PSV->isConstantPool();
+    return false;
+  };
+  // Load/store operations may access the stack indirectly when we previously
+  // computed an address to a stack location.
+  if (StackAddressUsed && MI.mayLoadOrStore() &&
+      (MI.isCall() || MI.hasUnmodeledSideEffects() || MI.memoperands_empty() ||
+       !all_of(MI.memoperands(), IsKnownNonStackPtr)))
+    return true;
+
+  if (MI.getOpcode() == FrameSetupOpcode ||
+      MI.getOpcode() == FrameDestroyOpcode) {
+    LLVM_DEBUG(dbgs() << "Frame instruction: " << MI << '\n');
+    return true;
+  }
+
+  if (MI.isDebugValue())
+    return false;
+
+  const auto &Ops = MI.operands();
+
+  auto FIOpIt = std::find_if(Ops.begin(), Ops.end(),
+                             [](const auto &MO) { return MO.isFI(); });
+  if (FIOpIt == Ops.end())
+    return false;
+
+  LLVM_DEBUG(dbgs() << "Use or define FI( " << FIOpIt->isFI() << "): " << MI
+                    << '\n');
+
+  return true;
+}
+
+/// calculateSets - collect the CSRs used in this function, compute
+/// the DF sets that describe the initial minimal regions in the
+/// Machine CFG around which CSR spills and restores must be placed.
+///
+/// Additionally, this function decides if shrink wrapping should
+/// be disabled for the current function, checking the following:
+///  1. the current function has more than 500 MBBs: heuristic limit
+///     on function size to reduce compile time impact of the current
+///     iterative algorithm.
+///  2. all CSRs are used in the entry block.
+///  3. all CSRs are used in all immediate successors of the entry block.
+///  4. all CSRs are used in a subset of blocks, each of which dominates
+///     all return blocks. These blocks, taken as a subgraph of the MCFG,
+///     are equivalent to the entry block since all execution paths pass
+///     through them.
+///
+bool ShrinkWrappingImpl::calculateSets(
+    MachineFunction &MF,
+    const ReversePostOrderTraversal<MachineBasicBlock *> &RPOT,
+    RegScavenger *RS) {
+  bool ShrinkWrappingEnabled = true;
+
+  // Sets used to compute spill, restore placement sets.
+  UsedCSRegs = getCurrentCSRsBitVec(RS);
+
+  SetOfRegs CurrentCSRs;
+  for (unsigned Reg : UsedCSRegs) {
+    CurrentCSRs.insert((unsigned)Reg);
+  }
+
+  findFastExitPath();
+
+  // If no CSRs used, we are done.
+  if (UsedCSRegs.empty()) {
+    LLVM_DEBUG(dbgs() << "DISABLED: " << MF.getName()
+                      << ": uses no callee-saved registers\n");
+    return false;
+  }
+
+  // Limit shrink wrapping via the current iterative bit vector
+  // implementation to functions with <= 500 MBBs.
+  if (MF.size() > 500) {
+    LLVM_DEBUG(dbgs() << "DISABLED: " << MF.getName() << ": too large ("
+                      << MF.size() << " MBBs)\n");
+    return false;
+  }
+
+  // Walk instructions in all MBBs, create CSRUsed[] sets, choose
+  // whether or not to shrink wrap this function.
+  // MachineLoopInfo &LI = getAnalysis<MachineLoopInfo>();
+  // MachineDominatorTree &DT = getAnalysis<MachineDominatorTree>();
+
+  for (MachineFunction::iterator MBB = MF.begin(), E = MF.end(); MBB != E;
+       ++MBB)
+    if (MBB->isReturnBlock())
+      ReturnBlocks.push_back(&(*MBB));
+
+  CSRegSet RegsAccessed;
+  bool allCSRUsesInEntryBlock = true;
+  for (MachineFunction::iterator MBBI = MF.begin(), MBBE = MF.end();
+       MBBI != MBBE; ++MBBI) {
+    MachineBasicBlock *MBB = &(*MBBI);
+    bool StackAddressUsed = false;
+    // Check if we found any stack accesses in the predecessors. We are not
+    // doing a full dataflow analysis here to keep things simple but just
+    // rely on a reverse portorder traversal (RPOT) to guarantee predecessors
+    // are already processed except for loops (and accept the conservative
+    // result for loops).
+    for (const MachineBasicBlock *Pred : MBB->predecessors()) {
+      if (StackAddressUsedBlockInfo.test(Pred->getNumber())) {
+        StackAddressUsed = true;
+        break;
+      }
+    }
+    for (MachineBasicBlock::iterator MI = MBB->begin(); MI != MBB->end();
+         ++MI) {
+      RegsAccessed.clear();
+      if (useOrDefFI(*MI, RS, StackAddressUsed)) {
+        SaveBlocks.push_back(MBB);
+        RestoreBlocks.push_back(MBB);
+        StackAddressUsed = true;
+      }
+      if (useOrDefCSR(*MI, RS, &RegsAccessed)) {
+        CSRUsed[MBB] |= RegsAccessed;
+        StackAddressUsed = true;
+        if (MBB != Entry && !CSRUsed[MBB].empty())
+          allCSRUsesInEntryBlock = false;
+      }
+    }
+    StackAddressUsedBlockInfo[MBB->getNumber()] = StackAddressUsed;
+    if (CSRUsed[MBB].empty())
+      continue;
+
+    // Propagate CSRUsed[MBB] in loops
+    if (MachineLoop *LP = MLI->getLoopFor(MBB)) {
+      // Push uses from inside loop to its parent loops,
+      // or to all other MBBs in its loop.
+      if (LP->getLoopDepth() > 1) {
+        for (MachineLoop *PLP = LP->getParentLoop(); PLP;
+             PLP = PLP->getParentLoop()) {
+          propagateUsesAroundLoop(MBB, PLP);
+        }
+      } else {
+        propagateUsesAroundLoop(MBB, LP);
+      }
+    }
+  }
+
+  if (allCSRUsesInEntryBlock) {
+    LLVM_DEBUG(dbgs() << "DISABLED: " << MF.getName()
+                      << ": all CSRs used in EntryBlock\n");
+    ShrinkWrappingEnabled = false;
+  } else {
+    bool allCSRsUsedInEntryFanout = true;
+    for (MachineBasicBlock::succ_iterator SI = Entry->succ_begin(),
+                                          SE = Entry->succ_end();
+         SI != SE; ++SI) {
+      MachineBasicBlock *SUCC = *SI;
+      if (CSRUsed[SUCC] != UsedCSRegs)
+        allCSRsUsedInEntryFanout = false;
+    }
+    if (allCSRsUsedInEntryFanout) {
+      LLVM_DEBUG(dbgs() << "DISABLED: " << MF.getName()
+                        << ": all CSRs used in imm successors of EntryBlock\n");
+      ShrinkWrappingEnabled = false;
+    }
+  }
+
+  if (ShrinkWrappingEnabled) {
+    // Check if MBB uses CSRs and dominates all exit nodes.
+    // Such nodes are equiv. to the entry node w.r.t.
+    // CSR uses: every path through the function must
+    // pass through this node. If each CSR is used at least
+    // once by these nodes, shrink wrapping is disabled.
+    CSRegSet CSRUsedInChokePoints;
+    for (MachineFunction::iterator MBBI = MF.begin(), MBBE = MF.end();
+         MBBI != MBBE; ++MBBI) {
+      MachineBasicBlock *MBB = &(*MBBI);
+      if (MBB == Entry || CSRUsed[MBB].empty() || MBB->succ_size() < 1)
+        continue;
+      bool dominatesExitNodes = true;
+      for (unsigned ri = 0, re = ReturnBlocks.size(); ri != re; ++ri)
+        if (!MDT->dominates(MBB, ReturnBlocks[ri])) {
+          dominatesExitNodes = false;
+          break;
+        }
+      if (dominatesExitNodes) {
+        CSRUsedInChokePoints |= CSRUsed[MBB];
+        if (CSRUsedInChokePoints == UsedCSRegs) {
+          LLVM_DEBUG(dbgs() << "DISABLED: " << MF.getName()
+                            << ": all CSRs used in choke point(s) at "
+                            << getBasicBlockName(MBB) << "\n");
+          ShrinkWrappingEnabled = false;
+          break;
+        }
+      }
+    }
+  }
+
+  // Return now if we have decided not to apply shrink wrapping
+  // to the current function.
+  if (!ShrinkWrappingEnabled)
+    return false;
+
+  LLVM_DEBUG({
+    dbgs() << "ENABLED: " << MF.getName();
+    if (HasFastExitPath)
+      dbgs() << " (fast exit path)";
+    dbgs() << "\n";
+    if (ShrinkWrappingDebugging >= BasicInfo) {
+      dbgs() << "------------------------------"
+             << "-----------------------------\n";
+      dbgs() << "UsedCSRegs = " << stringifyCSRegSet(UsedCSRegs) << "\n";
+      if (ShrinkWrappingDebugging >= Details) {
+        dbgs() << "------------------------------"
+               << "-----------------------------\n";
+        dumpAllUsed();
+      }
+    }
+  });
+
+  // Build initial DF sets to determine minimal regions in the
+  // Machine CFG around which CSRs must be spilled and restored.
+  calculateAnticAvail(MF, RS);
+
+  return true;
+}
+
+/// calcSpillPlacements - determine which CSRs should be spilled
+/// in MBB using AnticIn sets of MBB's predecessors, keeping track
+/// of changes to spilled reg sets. Add MBB to the set of blocks
+/// that need to be processed for propagating use info to cover
+/// multi-entry/exit regions.
+///
+bool ShrinkWrappingImpl::calcSpillPlacements(
+    AuxGraphNode *Node, SmallVectorImpl<AuxGraphNode *> &blks,
+    CSRegNodeMap &prevSpills) {
+  bool placedSpills = false;
+  // Intersect (CSRegs - AnticIn[P]) for P in Predecessors(MBB)
+  CSRegSet anticInPreds;
+  SmallVector<AuxGraphNode *, 4> predecessors;
+  for (auto PI = Node->pred_begin(), PE = Node->pred_end(); PI != PE; ++PI) {
+    AuxGraphNode *PRED = *PI;
+    if (PRED != Node)
+      predecessors.push_back(PRED);
+  }
+  unsigned i = 0, e = predecessors.size();
+  if (i != e) {
+    AuxGraphNode *PRED = predecessors[i];
+    anticInPreds = UsedCSRegs - AnticIn[PRED];
+    for (++i; i != e; ++i) {
+      PRED = predecessors[i];
+      anticInPreds &= (UsedCSRegs - AnticIn[PRED]);
+    }
+  } else {
+    // Handle uses in entry blocks (which have no predecessors).
+    // This is necessary because the DFA formulation assumes the
+    // entry and (multiple) exit nodes cannot have CSR uses, which
+    // is not the case in the real world.
+    anticInPreds = UsedCSRegs;
+  }
+  // Compute spills required at MBB:
+  CSRSave[Node] |= (AnticIn[Node] - AvailIn[Node]) & anticInPreds;
+
+  if (!CSRSave[Node].empty()) {
+    if (Node == AuxillaryCFG.getNode(Entry)) {
+      for (unsigned ri = 0, re = ReturnBlocks.size(); ri != re; ++ri)
+        CSRRestore[AuxillaryCFG.getNode(ReturnBlocks[ri])] |= CSRSave[Node];
+    } else {
+      // Reset all regs spilled in MBB that are also spilled in EntryBlock.
+      if (CSRSave[AuxillaryCFG.getNode(Entry)].intersects(CSRSave[Node])) {
+        CSRSave[Node] = CSRSave[Node] - CSRSave[AuxillaryCFG.getNode(Entry)];
+      }
+    }
+  }
+  placedSpills = (CSRSave[Node] != prevSpills[Node]);
+  prevSpills[Node] = CSRSave[Node];
+  // Remember this block for adding restores to successor
+  // blocks for multi-entry region.
+  if (placedSpills)
+    blks.push_back(Node);
+
+  LLVM_DEBUG(if (!CSRSave[Node].empty() &&
+                 ShrinkWrappingDebugging >= Iterations) dbgs()
+             << "SAVE[" << Node->Name
+             << "] = " << stringifyCSRegSet(CSRSave[Node]) << "\n");
+
+  return placedSpills;
+}
+
+/// calcRestorePlacements - determine which CSRs should be restored
+/// in MBB using AvailOut sets of MBB's succcessors, keeping track
+/// of changes to restored reg sets. Add MBB to the set of blocks
+/// that need to be processed for propagating use info to cover
+/// multi-entry/exit regions.
+///
+bool ShrinkWrappingImpl::calcRestorePlacements(
+    AuxGraphNode *Node, SmallVectorImpl<AuxGraphNode *> &blks,
+    CSRegNodeMap &prevRestores) {
+  bool placedRestores = false;
+  // Intersect (CSRegs - AvailOut[S]) for S in Successors(MBB)
+  CSRegSet availOutSucc;
+  SmallVector<AuxGraphNode *, 4> successors;
+  for (auto SI = Node->succ_begin(), SE = Node->succ_end(); SI != SE; ++SI) {
+    AuxGraphNode *SUCC = *SI;
+    if (SUCC != Node)
+      successors.push_back(SUCC);
+  }
+  unsigned i = 0, e = successors.size();
+  if (i != e) {
+    AuxGraphNode *SUCC = successors[i];
+    availOutSucc = UsedCSRegs - AvailOut[SUCC];
+    for (++i; i != e; ++i) {
+      SUCC = successors[i];
+      availOutSucc &= (UsedCSRegs - AvailOut[SUCC]);
+    }
+  } else {
+    CSRegSet Used;
+    if (Node->MatchMBB)
+      Used = CSRUsed[Node->MatchMBB];
+    if (!Used.empty() || !AvailOut[Node].empty()) {
+      // Handle uses in return blocks (which have no successors).
+      // This is necessary because the DFA formulation assumes the
+      // entry and (multiple) exit nodes cannot have CSR uses, which
+      // is not the case in the real world.
+      availOutSucc = UsedCSRegs;
+    }
+  }
+  // Compute restores required at MBB:
+  CSRRestore[Node] |= (AvailOut[Node] - AnticOut[Node]) & availOutSucc;
+
+  // Postprocess restore placements at MBB.
+  // Remove the CSRs that are restored in the return blocks.
+  // Lest this be confusing, note that:
+  // CSRSave[EntryBlock] == CSRRestore[B] for all B in ReturnBlocks.
+  if (Node->succ_size() && !CSRRestore[Node].empty()) {
+    if (!CSRSave[AuxillaryCFG.getNode(Entry)].empty())
+      CSRRestore[Node] =
+          CSRRestore[Node] - CSRSave[AuxillaryCFG.getNode(Entry)];
+  }
+  placedRestores = (CSRRestore[Node] != prevRestores[Node]);
+  prevRestores[Node] = CSRRestore[Node];
+  // Remember this block for adding saves to predecessor
+  // blocks for multi-entry region.
+  if (placedRestores)
+    blks.push_back(Node);
+
+  LLVM_DEBUG(if (!CSRRestore[Node].empty() &&
+                 ShrinkWrappingDebugging >= Iterations) dbgs()
+             << "RESTORE[" << Node->Name
+             << "] = " << stringifyCSRegSet(CSRRestore[Node]) << "\n");
+
+  return placedRestores;
+}
+
+/// placeSpillsAndRestores - place spills and restores of CSRs
+/// used in MBBs in minimal regions that contain the uses.
+///
+void ShrinkWrappingImpl::placeSpillsAndRestores(MachineFunction &Fn) {
+  CSRegNodeMap prevCSRSave;
+  CSRegNodeMap prevCSRRestore;
+  SmallVector<AuxGraphNode *, 4> cvBlocks, ncvBlocks;
+  bool changed = true;
+  unsigned iterations = 0;
+
+  // Iterate computation of spill and restore placements in the MCFG until:
+  //   1. CSR use info has been fully propagated around the MCFG, and
+  //   2. computation of CSRSave[], CSRRestore[] reach fixed points.
+  while (changed) {
+    changed = false;
+    ++iterations;
+
+    LLVM_DEBUG(if (ShrinkWrappingDebugging >= Iterations) dbgs()
+               << "iter " << iterations
+               << " --------------------------------------------------\n");
+
+    // Calculate CSR{Save,Restore} sets using Antic, Avail on the MCFG,
+    // which determines the placements of spills and restores.
+    // Keep track of changes to spills, restores in each iteration to
+    // minimize the total iterations.
+    for (auto &Node : AuxillaryCFG.Nodes) {
+      // Place spills for CSRs in MBB.
+      calcSpillPlacements(Node, cvBlocks, prevCSRSave);
+
+      // Place restores for CSRs in MBB.
+      calcRestorePlacements(Node, cvBlocks, prevCSRRestore);
+    }
+  }
+
+  // Check for effectiveness:
+  //  SR0 = {r | r in CSRSave[EntryBlock], CSRRestore[RB], RB in ReturnBlocks}
+  //  numSRReduced = |(UsedCSRegs - SR0)|, approx. SR0 by CSRSave[EntryBlock]
+  // Gives a measure of how many CSR spills have been moved from EntryBlock
+  // to minimal regions enclosing their uses.
+  CSRegSet notSpilledInEntryBlock =
+      (UsedCSRegs - CSRSave[AuxillaryCFG.getNode(Entry)]);
+  unsigned numSRReducedThisFunc = notSpilledInEntryBlock.count();
+  numSRReduced += numSRReducedThisFunc;
+  LLVM_DEBUG(if (ShrinkWrappingDebugging >= BasicInfo) {
+    dbgs() << "-----------------------------------------------------------\n";
+    dbgs() << "total iterations = " << iterations << " ( " << Fn.getName()
+           << " " << numSRReducedThisFunc << " " << Fn.size() << " )\n";
+    dbgs() << "-----------------------------------------------------------\n";
+    dumpSRSets();
+    dbgs() << "-----------------------------------------------------------\n";
+  });
+}
+
+bool ShrinkWrappingImpl::run(MachineFunction &MF) {
+  LLVM_DEBUG(dbgs() << "**** Analysing " << MF.getName() << '\n');
+
+  const TargetRegisterInfo *TRI = MF.getSubtarget().getRegisterInfo();
+
+  std::unique_ptr<RegScavenger> RS(
+      TRI->requiresRegisterScavenging(MF) ? new RegScavenger() : nullptr);
+
+  init(MF, RS.get());
+
+  ReversePostOrderTraversal<MachineBasicBlock *> RPOT(&*MF.begin());
+  if (containsIrreducibleCFG<MachineBasicBlock *>(RPOT, *MLI)) {
+    // If MF is irreducible, a block may be in a loop without
+    // MachineLoopInfo reporting it. I.e., we may use the
+    // post-dominance property in loops, which lead to incorrect
+    // results. Moreover, we may miss that the prologue and
+    // epilogue are not in the same loop, leading to unbalanced
+    // construction/deconstruction of the stack frame.
+    return giveUpWithRemarks(ORE, "UnsupportedIrreducibleCFG",
+                             "Irreducible CFGs are not supported yet.",
+                             MF.getFunction().getSubprogram(), &MF.front());
+  }
+
+  MachineFrameInfo &MFI = MF.getFrameInfo();
+  if (MFI.hasVarSizedObjects()) {
+    LLVM_DEBUG(dbgs() << "Can't split save/restore points, because frame "
+                         "contains var sized objects\n");
+    return false;
+  }
+
+  for (MachineBasicBlock *MBB : RPOT) {
+    if (MBB->isEHFuncletEntry())
+      return giveUpWithRemarks(ORE, "UnsupportedEHFunclets",
+                               "EH Funclets are not supported yet.",
+                               MBB->front().getDebugLoc(), MBB);
+
+    if (MBB->isEHPad() || MBB->isInlineAsmBrIndirectTarget())
+      return giveUpWithRemarks(
+          ORE, "EHPads and isInlineAsmBrIndirectTargets",
+          "EHPads and isInlineAsmBrIndirectTargets are not supported yet.",
+          MBB->front().getDebugLoc(), MBB);
+  }
+
+  const MachineJumpTableInfo *MJTI = MachineFunc->getJumpTableInfo();
+  if (MJTI)
+    return giveUpWithRemarks(ORE, "UnsupportedMJTI",
+                             "JumpTables are not supported yet.",
+                             MF.getFunction().getSubprogram(), &MF.front());
+
+  createAuxillaryCFG();
+
+  // Initially, conservatively assume that stack addresses can be used in each
+  // basic block and change the state only for those basic blocks for which we
+  // were able to prove the opposite.
+  StackAddressUsedBlockInfo.resize(MF.getNumBlockIDs(), true);
+  bool HasCandidates = calculateSets(MF, RPOT, RS.get());
+  StackAddressUsedBlockInfo.clear();
+  if (!HasCandidates)
+    return false;
+
+  placeSpillsAndRestores(MF);
+  setupCFG();
+  verifySpillRestorePlacement();
+
+  setupSaveRestorePoints();
+  Prolog = SaveBlocks.empty()
+               ? nullptr
+               : MDT->findNearestCommonDominator(iterator_range(SaveBlocks));
+  Epilog =
+      RestoreBlocks.empty()
+          ? nullptr
+          : MPDT->findNearestCommonDominator(iterator_range(RestoreBlocks));
+
+  if (SavePoints.areMultiple() || RestorePoints.areMultiple()) {
+    ++NumFuncWithSplitting;
+  }
+
+  LLVM_DEBUG(dbgs() << "Final shrink wrap candidates:\n");
+
+  LLVM_DEBUG(dbgs() << "SavePoints:\n");
+  LLVM_DEBUG(SavePoints.dump(TRI));
+
+  LLVM_DEBUG(dbgs() << "RestorePoints:\n");
+  LLVM_DEBUG(RestorePoints.dump(TRI));
+
+  MFI.setProlog(Prolog);
+  MFI.setEpilog(Epilog);
+  if (!SavePoints.get().empty() && !RestorePoints.get().empty()) {
+    MFI.setSavePoints(SavePoints.get());
+    MFI.setRestorePoints(RestorePoints.get());
+  } else {
+    ++NumNotSaveOrRestore;
+  }
+  ++NumCandidates;
+
+  return true;
+}
+
+MachineBasicBlock *ShrinkWrappingImpl::splitEdge(MachineBasicBlock *Pred,
+                                                 MachineBasicBlock *Succ) {
+  MachineBasicBlock *NewBB = MachineFunc->CreateMachineBasicBlock();
+
+  MachineJumpTableInfo *MJTI = MachineFunc->getJumpTableInfo();
+  if (MJTI)
+    MJTI->ReplaceMBBInJumpTables(Succ, NewBB);
+
+  MachineBasicBlock *FallThrough = Pred->getFallThrough(false);
+  MachineFunc->insert(MachineFunc->end(), NewBB);
+  bool Done = false;
+  const TargetInstrInfo *TII = MachineFunc->getSubtarget().getInstrInfo();
+  if (Pred->getFirstTerminator() == Pred->end()) {
+    Done = true;
+    TII->insertUnconditionalBranch(*Pred, NewBB, DebugLoc());
+  } else {
+    for (auto &Term : Pred->terminators()) {
+      if (Term.isUnconditionalBranch() &&
+          (Term.getOperand(0).getMBB() == Succ)) {
+        Done = true;
+        Term.getOperand(0).setMBB(NewBB);
+      } else if (Term.isConditionalBranch()) {
+        for (auto &MO : Term.operands()) {
+          if (MO.isMBB() && (MO.getMBB() == Succ)) {
+            Done = true;
+            MO.setMBB(NewBB);
+          }
+        }
+      }
+    }
+  }
+
+  if (!Done && !Pred->isLayoutSuccessor(NewBB) && FallThrough == Succ) {
+    TII->insertUnconditionalBranch(*Pred, NewBB, DebugLoc());
+  }
+
+  // TODO: switch
+
+  Pred->replaceSuccessor(Succ, NewBB); // Remove old edge
+  NewBB->addSuccessor(Succ);           // Add NewBB → Succ
+
+  for (const MachineBasicBlock::RegisterMaskPair &LI : Succ->liveins())
+    NewBB->addLiveIn(LI.PhysReg);
+
+  TII->insertUnconditionalBranch(*NewBB, Succ, DebugLoc());
+
+  return NewBB;
+}
+
+void ShrinkWrappingImpl::setupCFG() {
+  for (auto &[Node, Regs] : CSRSave) {
+    if (!Regs.empty() && !Node->MatchMBB) {
+      assert(Node->pred_size() == 1 &&
+             "Auxillary node can have only one predecessor!");
+      assert(Node->succ_size() == 1 &&
+             "Auxillary node can have only one successor!");
+      MachineBasicBlock *NewBB = splitEdge((*Node->pred_begin())->MatchMBB,
+                                           (*Node->succ_begin())->MatchMBB);
+      Node->MatchMBB = NewBB;
+    }
+  }
+
+  for (auto &[Node, Regs] : CSRRestore) {
+    if (!Regs.empty() && !Node->MatchMBB) {
+      assert(Node->pred_size() == 1 &&
+             "Auxillary node can have only one predecessor!");
+      assert(Node->succ_size() == 1 &&
+             "Auxillary node can have only one successor!");
+      MachineBasicBlock *NewBB = splitEdge((*Node->pred_begin())->MatchMBB,
+                                           (*Node->succ_begin())->MatchMBB);
+      Node->MatchMBB = NewBB;
+    }
+  }
+}
+
+void ShrinkWrappingImpl::setupSaveRestorePoints() {
+  for (auto &[Node, Regs] : CSRSave) {
+    for (Register Reg : Regs)
+      SavePoints.insertReg(Reg, Node->MatchMBB, SaveBlocks);
+  }
+  for (auto &[Node, Regs] : CSRRestore) {
+    for (Register Reg : Regs)
+      RestorePoints.insertReg(Reg, Node->MatchMBB, RestoreBlocks);
+  }
+}
+
+bool ShrinkWrappingLegacy::runOnMachineFunction(MachineFunction &MF) {
+  if (skipFunction(MF.getFunction()) || MF.empty() ||
+      !ShrinkWrappingImpl::isShrinkWrappingEnabled(MF))
+    return false;
+
+  MachineDominatorTree *MDT =
+      &getAnalysis<MachineDominatorTreeWrapperPass>().getDomTree();
+  MachinePostDominatorTree *MPDT =
+      &getAnalysis<MachinePostDominatorTreeWrapperPass>().getPostDomTree();
+  MachineBlockFrequencyInfo *MBFI =
+      &getAnalysis<MachineBlockFrequencyInfoWrapperPass>().getMBFI();
+  MachineLoopInfo *MLI = &getAnalysis<MachineLoopInfoWrapperPass>().getLI();
+  MachineOptimizationRemarkEmitter *ORE =
+      &getAnalysis<MachineOptimizationRemarkEmitterPass>().getORE();
+
+  return ShrinkWrappingImpl(MDT, MPDT, MBFI, MLI, ORE).run(MF);
+}
+
+PreservedAnalyses
+ShrinkWrappingPass::run(MachineFunction &MF,
+                        MachineFunctionAnalysisManager &MFAM) {
+  MFPropsModifier _(*this, MF);
+  if (MF.empty() || !ShrinkWrappingImpl::isShrinkWrappingEnabled(MF))
+    return PreservedAnalyses::all();
+
+  MachineDominatorTree &MDT = MFAM.getResult<MachineDominatorTreeAnalysis>(MF);
+  MachinePostDominatorTree &MPDT =
+      MFAM.getResult<MachinePostDominatorTreeAnalysis>(MF);
+  MachineBlockFrequencyInfo &MBFI =
+      MFAM.getResult<MachineBlockFrequencyAnalysis>(MF);
+  MachineLoopInfo &MLI = MFAM.getResult<MachineLoopAnalysis>(MF);
+  MachineOptimizationRemarkEmitter &ORE =
+      MFAM.getResult<MachineOptimizationRemarkEmitterAnalysis>(MF);
+
+  ShrinkWrappingImpl(&MDT, &MPDT, &MBFI, &MLI, &ORE).run(MF);
+  return PreservedAnalyses::all();
+}
+
+bool ShrinkWrappingImpl::isShrinkWrappingEnabled(const MachineFunction &MF) {
+  const TargetFrameLowering *TFI = MF.getSubtarget().getFrameLowering();
+
+  switch (EnableShrinkWrappingOpt) {
+  case cl::boolOrDefault::BOU_UNSET:
+    return TFI->enableShrinkWrapping(MF) &&
+           // Windows with CFI has some limitations that make it impossible
+           // to use shrink-wrapping.
+           !MF.getTarget().getMCAsmInfo().usesWindowsCFI() &&
+           // Sanitizers look at the value of the stack at the location
+           // of the crash. Since a crash can happen anywhere, the
+           // frame must be lowered before anything else happen for the
+           // sanitizers to be able to get a correct stack frame.
+           !(MF.getFunction().hasFnAttribute(Attribute::SanitizeAddress) ||
+             MF.getFunction().hasFnAttribute(Attribute::SanitizeThread) ||
+             MF.getFunction().hasFnAttribute(Attribute::SanitizeMemory) ||
+             MF.getFunction().hasFnAttribute(Attribute::SanitizeType) ||
+             MF.getFunction().hasFnAttribute(Attribute::SanitizeHWAddress));
+  // If EnableShrinkWrap is set, it takes precedence on whatever the
+  // target sets. The rational is that we assume we want to test
+  // something related to shrink-wrapping.
+  case cl::boolOrDefault::BOU_TRUE:
+    return true;
+  case cl::boolOrDefault::BOU_FALSE:
+    return false;
+  }
+  llvm_unreachable("Invalid shrink-wrapping state");
+}
diff --git a/llvm/lib/Passes/PassBuilder.cpp b/llvm/lib/Passes/PassBuilder.cpp
index 858c76706427e8..5527935239c018 100644
--- a/llvm/lib/Passes/PassBuilder.cpp
+++ b/llvm/lib/Passes/PassBuilder.cpp
@@ -191,6 +191,7 @@
 #include "llvm/CodeGen/SelectOptimize.h"
 #include "llvm/CodeGen/ShadowStackGCLowering.h"
 #include "llvm/CodeGen/ShrinkWrap.h"
+#include "llvm/CodeGen/ShrinkWrapping.h"
 #include "llvm/CodeGen/SjLjEHPrepare.h"
 #include "llvm/CodeGen/SlotIndexes.h"
 #include "llvm/CodeGen/SpillPlacement.h"
diff --git a/llvm/lib/Target/AMDGPU/SILowerSGPRSpills.cpp b/llvm/lib/Target/AMDGPU/SILowerSGPRSpills.cpp
index d0a57e163d29a0..0d3e8fec70afcb 100644
--- a/llvm/lib/Target/AMDGPU/SILowerSGPRSpills.cpp
+++ b/llvm/lib/Target/AMDGPU/SILowerSGPRSpills.cpp
@@ -193,20 +193,21 @@ void SILowerSGPRSpills::calculateSaveRestoreBlocks(MachineFunction &MF) {
   // So set the save points for those.
 
   // Use the points found by shrink-wrapping, if any.
+
   if (!MFI.getSavePoints().empty()) {
-    assert(MFI.getSavePoints().size() == 1 &&
-           "Multiple save points not yet supported!");
-    const auto &SavePoint = *MFI.getSavePoints().begin();
-    SaveBlocks.push_back(SavePoint.first);
-    assert(MFI.getRestorePoints().size() == 1 &&
-           "Multiple restore points not yet supported!");
-    const auto &RestorePoint = *MFI.getRestorePoints().begin();
-    MachineBasicBlock *RestoreBlock = RestorePoint.first;
-    // If RestoreBlock does not have any successor and is not a return block
-    // then the end point is unreachable and we do not need to insert any
-    // epilogue.
-    if (!RestoreBlock->succ_empty() || RestoreBlock->isReturnBlock())
-      RestoreBlocks.push_back(RestoreBlock);
+    assert(!MFI.getRestorePoints().empty() &&
+           "Both restores and saves must be set");
+    for (auto &item : MFI.getSavePoints())
+      SaveBlocks.push_back(item.first);
+
+    for (auto &item : MFI.getRestorePoints()) {
+      MachineBasicBlock *RestoreBlock = item.first;
+      // If RestoreBlock does not have any successor and is not a return block
+      // then the end point is unreachable and we do not need to insert any
+      // epilogue.
+      if (!RestoreBlock->succ_empty() || RestoreBlock->isReturnBlock())
+        RestoreBlocks.push_back(RestoreBlock);
+    }
     return;
   }
 
diff --git a/llvm/lib/Target/PowerPC/PPCFrameLowering.cpp b/llvm/lib/Target/PowerPC/PPCFrameLowering.cpp
index 6d8300fc8faa26..fbdb66bed7adb3 100644
--- a/llvm/lib/Target/PowerPC/PPCFrameLowering.cpp
+++ b/llvm/lib/Target/PowerPC/PPCFrameLowering.cpp
@@ -2094,8 +2094,6 @@ void PPCFrameLowering::processFunctionBeforeFrameFinalized(MachineFunction &MF,
   // won't be generated by emitEpilogue(), because shrink-wrap has chosen new
   // RestoreBlock. So we handle this case here.
   if (!MFI.getSavePoints().empty() && MFI.hasTailCall()) {
-    assert(MFI.getRestorePoints().size() < 2 &&
-           "MFI can't contain multiple restore points!");
     for (MachineBasicBlock &MBB : MF) {
       if (MBB.isReturnBlock() && (!MFI.getRestorePoints().contains(&MBB)))
         createTailCallBranchInstr(MBB);
diff --git a/llvm/test/CodeGen/AArch64/live-debugvalues-sve.mir b/llvm/test/CodeGen/AArch64/live-debugvalues-sve.mir
index fa1da262fbd39e..2e41f3edd5da19 100644
--- a/llvm/test/CodeGen/AArch64/live-debugvalues-sve.mir
+++ b/llvm/test/CodeGen/AArch64/live-debugvalues-sve.mir
@@ -121,9 +121,17 @@ frameInfo:
   hasCalls:        true
   maxCallFrameSize: 0
   savePoint:
-  - point: '%bb.1'
+    - point:           '%bb.1'
+      registers:
+        - '$fp'
+        - '$lr'
+        - '$x28'
   restorePoint:
-  - point: '%bb.1'
+    - point:           '%bb.1'
+      registers:
+        - '$fp'
+        - '$lr'
+        - '$x28'
 stack:
   - { id: 0, size: 16, alignment: 16, stack-id: scalable-vector }
 machineFunctionInfo: {}
diff --git a/llvm/tools/llvm-reduce/ReducerWorkItem.cpp b/llvm/tools/llvm-reduce/ReducerWorkItem.cpp
index 6b202aec3e8f85..5e242140a0016b 100644
--- a/llvm/tools/llvm-reduce/ReducerWorkItem.cpp
+++ b/llvm/tools/llvm-reduce/ReducerWorkItem.cpp
@@ -103,15 +103,9 @@ static void cloneFrameInfo(
   DstMFI.setCVBytesOfCalleeSavedRegisters(
       SrcMFI.getCVBytesOfCalleeSavedRegisters());
 
-  assert(SrcMFI.getSavePoints().size() < 2 &&
-         "Multiple restore points not yet supported!");
-
   DstMFI.setSavePoints(
       constructSaveRestorePoints(SrcMFI.getSavePoints(), Src2DstMBB));
 
-  assert(SrcMFI.getRestorePoints().size() < 2 &&
-         "Multiple restore points not yet supported!");
-
   DstMFI.setRestorePoints(
       constructSaveRestorePoints(SrcMFI.getRestorePoints(), Src2DstMBB));
 

>From 16fb4f03db1bfc29c2e0747f81050db2db5103da Mon Sep 17 00:00:00 2001
From: Elizaveta Noskova <elizaveta.noskova at syntacore.com>
Date: Mon, 28 Sep 2026 15:28:48 +0300
Subject: [PATCH 2/2] [draft][llvm] RISCV: enable multiple save/restore points
 in shrink wrapping

Implement the RISCV target hooks and frame lowering support for the data-flow ShrinkWrapping pass:
 - per-save/restore-point callee-saved register info,
 - callee-saved frame index offset adjustment,
 - CFI handling needed when save/restore points are split away from the prologue/epilogue.

Add a shrinkwrap-split.mir test exercising the pass on RISCV.
---
 llvm/lib/Target/RISCV/RISCVFrameLowering.cpp | 187 ++++++++++--
 llvm/lib/Target/RISCV/RISCVFrameLowering.h   |   7 +
 llvm/lib/Target/RISCV/RISCVRegisterInfo.cpp  |  29 +-
 llvm/lib/Target/RISCV/RISCVRegisterInfo.h    |   3 +
 llvm/test/CodeGen/RISCV/shrinkwrap-split.mir | 284 +++++++++++++++++++
 5 files changed, 488 insertions(+), 22 deletions(-)
 create mode 100644 llvm/test/CodeGen/RISCV/shrinkwrap-split.mir

diff --git a/llvm/lib/Target/RISCV/RISCVFrameLowering.cpp b/llvm/lib/Target/RISCV/RISCVFrameLowering.cpp
index 4732b4153aae63..ad7fa55413d19e 100644
--- a/llvm/lib/Target/RISCV/RISCVFrameLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVFrameLowering.cpp
@@ -572,10 +572,9 @@ uint64_t RISCVFrameLowering::getStackSizeWithRVVPadding(
   return alignTo(MFI.getStackSize() + RVFI->getRVVPadding(), getStackAlign());
 }
 
-static SmallVector<CalleeSavedInfo, 8>
-getUnmanagedCSI(const MachineFunction &MF,
-                const std::vector<CalleeSavedInfo> &CSI,
-                bool ReverseOrder = false) {
+SmallVector<CalleeSavedInfo, 8> RISCVFrameLowering::getUnmanagedCSI(
+    const MachineFunction &MF, const std::vector<CalleeSavedInfo> &CSI,
+    bool ReverseOrder) const {
   const MachineFrameInfo &MFI = MF.getFrameInfo();
   SmallVector<CalleeSavedInfo, 8> NonLibcallCSI;
 
@@ -598,7 +597,6 @@ getRVVCalleeSavedInfo(const MachineFunction &MF,
                       const std::vector<CalleeSavedInfo> &CSI) {
   const MachineFrameInfo &MFI = MF.getFrameInfo();
   SmallVector<CalleeSavedInfo, 8> RVVCSI;
-
   for (auto &CS : CSI) {
     int FI = CS.getFrameIdx();
     if (FI >= 0 && MFI.getStackID(FI) == TargetStackID::ScalableVector)
@@ -1020,17 +1018,33 @@ void RISCVFrameLowering::emitPrologue(MachineFunction &MF,
   // Determine the correct frame layout
   determineFrameLayout(MF);
 
-  const auto &CSI = MFI.getCalleeSavedInfo();
+  const auto &CSI = MFI.getSaveCSInfo(&MBB);
 
   // Skip to before the spills of scalar callee-saved registers
   // FIXME: assumes exactly one instruction is used to restore each
   // callee-saved register.
-  MBBI =
-      std::prev(MBBI, getRVVCalleeSavedInfo(MF, CSI).size() +
-                          getUnmanagedCSI(MF, CSI, PreferAscendingLS).size());
-  CFIInstBuilder CFIBuilder(MBB, MBBI, MachineInstr::FrameSetup);
   bool NeedsDwarfCFI = needsDwarfCFI(MF);
 
+  // For Zilsd on RV32, append GPRPair registers to the CSR list. This prevents
+  // the need to create register sets for each abi which is a lot more complex.
+  // Don't use Zilsd for callee-saved coalescing if the required alignment
+  // exceeds the stack alignment.
+  bool UseZilsd = !STI.is64Bit() && STI.hasStdExtZilsd() &&
+                  STI.getZilsdAlign() <= getStackAlign();
+
+  // For scalar register spills we skip 2 instrs at once, because right after
+  // spills there are cfi instructions. At the moment of prolog emission they
+  // are already inserted for scalar instructions, but not for vector
+  // instructions.
+  int ScalarDistance = getUnmanagedCSI(MF, CSI).size();
+
+  if (NeedsDwarfCFI && !UseZilsd)
+    ScalarDistance *= 2;
+
+  int VectorDistance = getRVVCalleeSavedInfo(MF, CSI).size();
+  MBBI = std::prev(MBBI, VectorDistance + ScalarDistance);
+
+  CFIInstBuilder CFIBuilder(MBB, MBBI, MachineInstr::FrameSetup);
   // If libcalls are used to spill and restore callee-saved registers, the frame
   // has two sections; the opaque section managed by the libcalls, and the
   // section managed by MachineFrameInfo which can also hold callee saved
@@ -1146,12 +1160,14 @@ void RISCVFrameLowering::emitPrologue(MachineFunction &MF,
   // to the stack, not before.
   // FIXME: assumes exactly one instruction is used to save each callee-saved
   // register.
-  std::advance(MBBI, getUnmanagedCSI(MF, CSI, PreferAscendingLS).size());
-  CFIBuilder.setInsertPoint(MBBI);
+  std::advance(MBBI, ScalarDistance);
+  if (UseZilsd)
+    CFIBuilder.setInsertPoint(MBBI);
 
   // Iterate over list of callee-saved registers and emit .cfi_offset
-  // directives.
-  if (NeedsDwarfCFI) {
+  // directives. This is only needed for Zilsd: otherwise the offsets are
+  // emitted together with the spills in spillCalleeSavedRegisters.
+  if (NeedsDwarfCFI && UseZilsd) {
     for (const CalleeSavedInfo &CS :
          getUnmanagedCSI(MF, CSI, PreferAscendingLS)) {
       MCRegister Reg = CS.getReg();
@@ -1186,8 +1202,10 @@ void RISCVFrameLowering::emitPrologue(MachineFunction &MF,
           MachineInstr::FrameSetup, getStackAlign());
     }
 
-    if (NeedsDwarfCFI)
+    if (NeedsDwarfCFI) {
+      CFIBuilder.setInsertPoint(MBBI);
       CFIBuilder.buildDefCFA(FPReg, RVFI->getVarArgsSaveSize());
+    }
   }
 
   uint64_t SecondSPAdjustAmount = 0;
@@ -1217,6 +1235,7 @@ void RISCVFrameLowering::emitPrologue(MachineFunction &MF,
     }
 
     if (NeedsDwarfCFI && !hasFP(MF)) {
+      CFIBuilder.setInsertPoint(MBBI);
       // Emit .cfi_def_cfa_expression "sp + StackSize + RVVStackSize * vlenb".
       CFIBuilder.insertCFIInst(createDefCFAExpression(
           *RI, SPReg,
@@ -1326,7 +1345,7 @@ void RISCVFrameLowering::emitEpilogue(MachineFunction &MF,
       --MBBI;
   }
 
-  const auto &CSI = MFI.getCalleeSavedInfo();
+  const auto &CSI = MFI.getRestoreCSInfo(&MBB);
 
   // Skip to before the restores of scalar callee-saved registers
   // FIXME: assumes exactly one instruction is used to restore each
@@ -1400,12 +1419,25 @@ void RISCVFrameLowering::emitEpilogue(MachineFunction &MF,
   if (NeedsDwarfCFI && hasFP(MF))
     CFIBuilder.buildDefCFA(SPReg, RealStackSize);
 
+  // For Zilsd on RV32, append GPRPair registers to the CSR list. This prevents
+  // the need to create register sets for each abi which is a lot more complex.
+  // Don't use Zilsd for callee-saved coalescing if the required alignment
+  // exceeds the stack alignment.
+  bool UseZilsd = !STI.is64Bit() && STI.hasStdExtZilsd() &&
+                  STI.getZilsdAlign() <= getStackAlign();
+
   // Skip to after the restores of scalar callee-saved registers
   // FIXME: assumes exactly one instruction is used to restore each
   // callee-saved register.
-  MBBI = std::next(FirstScalarCSRRestoreInsn,
-                   getUnmanagedCSI(MF, CSI, PreferAscendingLS).size());
-  CFIBuilder.setInsertPoint(MBBI);
+  // Skip CSR restore instructions + corresponding cfi restore instructions
+  int ScalarDistance = getUnmanagedCSI(MF, CSI).size();
+
+  if (NeedsDwarfCFI && !UseZilsd)
+    ScalarDistance *= 2;
+
+  MBBI = std::next(FirstScalarCSRRestoreInsn, ScalarDistance);
+  if (UseZilsd)
+    CFIBuilder.setInsertPoint(MBBI);
 
   if (getLibCallID(MF, CSI) != -1) {
     // tail __riscv_restore_[0-12] instruction is considered as a terminator,
@@ -1420,8 +1452,10 @@ void RISCVFrameLowering::emitEpilogue(MachineFunction &MF,
     return;
   }
 
-  // Recover callee-saved registers.
-  if (NeedsDwarfCFI) {
+  // Recover callee-saved registers. This is only needed for Zilsd: otherwise
+  // the CFI restore instructions are emitted together with the restores in
+  // restoreCalleeSavedRegisters.
+  if (NeedsDwarfCFI && UseZilsd) {
     for (const CalleeSavedInfo &CS :
          getUnmanagedCSI(MF, CSI, PreferAscendingLS)) {
       MCRegister Reg = CS.getReg();
@@ -2431,6 +2465,34 @@ bool RISCVFrameLowering::assignCalleeSavedSpillSlots(
   return true;
 }
 
+static int64_t calculateCSRSpillOffsets(MachineFrameInfo &MFI,
+                                        const TargetFrameLowering *TFI,
+                                        int MinCSFI, int FrameIdx) {
+  int LocalAreaOffset = -TFI->getOffsetOfLocalArea();
+  Align MaxAlign = MFI.getMaxAlign();
+  Align Alignment = MFI.getObjectAlign(FrameIdx);
+  MaxAlign = std::max(MaxAlign, Alignment);
+  int64_t Offset = LocalAreaOffset;
+
+  for (int i = MFI.getObjectIndexBegin(); i != 0; ++i) {
+    // Only allocate objects on the default stack.
+    if (MFI.getStackID(i) != TargetStackID::Default)
+      continue;
+
+    int64_t FixedOff;
+    FixedOff = -MFI.getObjectOffset(i);
+    if (FixedOff > Offset)
+      Offset = FixedOff;
+  }
+
+  for (int i = MinCSFI; i <= FrameIdx; ++i) {
+    Offset += MFI.getObjectSize(i);
+  }
+
+  Offset = alignTo(Offset, Alignment);
+  return -Offset;
+}
+
 bool RISCVFrameLowering::spillCalleeSavedRegisters(
     MachineBasicBlock &MBB, MachineBasicBlock::iterator MI,
     ArrayRef<CalleeSavedInfo> CSI, const TargetRegisterInfo *TRI) const {
@@ -2458,6 +2520,7 @@ bool RISCVFrameLowering::spillCalleeSavedRegisters(
       MBB.addLiveIn(Reg);
   }
 
+  // Emit CM.PUSH with base SPimm & evaluate Push stack
   if (RVFI->isPushable(*MF)) {
     // Emit CM.PUSH with base StackAdj & evaluate Push stack
     unsigned PushedRegNum = RVFI->getRVPushRegs();
@@ -2504,7 +2567,61 @@ bool RISCVFrameLowering::spillCalleeSavedRegisters(
                               MachineInstr::FrameSetup);
     }
   };
+
   storeRegsToStackSlots(UnmanagedCSI);
+
+  bool NeedsDwarfCFI = needsDwarfCFI(*MF);
+  // For Zilsd on RV32, append GPRPair registers to the CSR list. This prevents
+  // the need to create register sets for each abi which is a lot more complex.
+  // Don't use Zilsd for callee-saved coalescing if the required alignment
+  // exceeds the stack alignment.
+  bool UseZilsd = !STI.is64Bit() && STI.hasStdExtZilsd() &&
+                  STI.getZilsdAlign() <= getStackAlign();
+  // Iterate over list of callee-saved registers and emit .cfi_offset
+  // directives.
+  if (NeedsDwarfCFI && !UseZilsd) {
+    CFIInstBuilder CFIBuilder(MBB, MI, MachineInstr::FrameSetup);
+    MachineFrameInfo &MFI = MF->getFrameInfo();
+
+    for (const CalleeSavedInfo &CS : UnmanagedCSI) {
+      int FrameIdx = CS.getFrameIdx();
+      if (FrameIdx < 0 ||
+          MFI.getStackID(FrameIdx) != TargetStackID::ScalableVector) {
+        int64_t Offset = 0;
+
+        auto *RVFI = MF->getInfo<RISCVMachineFunctionInfo>();
+        std::vector<CalleeSavedInfo> GCSI = MFI.getCalleeSavedInfo();
+        unsigned MinCSFI = std::numeric_limits<unsigned>::max();
+        for (auto CS : GCSI) {
+          unsigned NonNegCSFI =
+              CS.getFrameIdx() >= 0 ? CS.getFrameIdx() : MinCSFI;
+          if (NonNegCSFI < MinCSFI)
+            MinCSFI = NonNegCSFI;
+        }
+        if (MinCSFI == std::numeric_limits<unsigned>::max())
+          MinCSFI = 0;
+
+        if (RVFI->isSiFivePreemptibleInterrupt(*MF)) {
+          for (int I = 0; I < 2; ++I) {
+            int FI = RVFI->getInterruptCSRFrameIndex(I);
+            MinCSFI = std::min<unsigned>(MinCSFI, FI);
+          }
+        }
+
+        if (FrameIdx < 0 &&
+            (RVFI->isPushable(*MF) || RVFI->useSaveRestoreLibCalls(*MF))) {
+          Offset = MFI.getObjectOffset(FrameIdx);
+        } else {
+          const TargetFrameLowering *TFI =
+              MF->getSubtarget().getFrameLowering();
+          Offset = calculateCSRSpillOffsets(MFI, TFI, MinCSFI, FrameIdx);
+        }
+        MCRegister Reg = CS.getReg();
+        CFIBuilder.buildOffset(Reg, Offset);
+      }
+    }
+  }
+
   storeRegsToStackSlots(RVVCSI);
 
   return true;
@@ -2600,8 +2717,25 @@ bool RISCVFrameLowering::restoreCalleeSavedRegisters(
     }
   };
   loadRegFromStackSlot(RVVCSI);
+
   loadRegFromStackSlot(UnmanagedCSI);
 
+  bool NeedsDwarfCFI = needsDwarfCFI(*MF);
+  // For Zilsd on RV32, append GPRPair registers to the CSR list. This prevents
+  // the need to create register sets for each abi which is a lot more complex.
+  // Don't use Zilsd for callee-saved coalescing if the required alignment
+  // exceeds the stack alignment.
+  bool UseZilsd = !STI.is64Bit() && STI.hasStdExtZilsd() &&
+                  STI.getZilsdAlign() <= getStackAlign();
+  // Recover callee-saved registers.
+  if (NeedsDwarfCFI && !UseZilsd) {
+    CFIInstBuilder CFIBuilder(MBB, MI, MachineInstr::FrameDestroy);
+    for (const CalleeSavedInfo &CS : UnmanagedCSI) {
+      MCRegister Reg = CS.getReg();
+      CFIBuilder.buildRestore(Reg);
+    }
+  }
+
   RISCVMachineFunctionInfo *RVFI = MF->getInfo<RISCVMachineFunctionInfo>();
   if (RVFI->useQCIInterrupt(*MF)) {
     // Don't emit anything here because restoration is handled by
@@ -2870,3 +3004,14 @@ RISCVFrameLowering::getInitialCFARegister(const MachineFunction &MF) const {
 uint64_t RISCVFrameLowering::getStackThreshold() const {
   return STI.is64Bit() ? INT64_MAX : UINT32_MAX;
 }
+
+bool RISCVFrameLowering::enableCSRSaveRestorePointsSplit() const {
+  // Zcmp extention introduces cm.push and cm.pop instructions, which allow to
+  // perform all spills and restores in one corresponding instruction. This
+  // contradicts the idea of splitting Save Restore points. "-msave-restore"
+  // does the same, not via new instructions but via save/restore libcalls.
+  if (!STI.hasStdExtZcmp() && !STI.enableSaveRestore() &&
+      !STI.hasVendorXqccmp())
+    return true;
+  return false;
+}
diff --git a/llvm/lib/Target/RISCV/RISCVFrameLowering.h b/llvm/lib/Target/RISCV/RISCVFrameLowering.h
index 5157c456a92c75..d86a73f0743312 100644
--- a/llvm/lib/Target/RISCV/RISCVFrameLowering.h
+++ b/llvm/lib/Target/RISCV/RISCVFrameLowering.h
@@ -31,6 +31,11 @@ class RISCVFrameLowering : public TargetFrameLowering {
 
   uint64_t getStackSizeWithRVVPadding(const MachineFunction &MF) const;
 
+  SmallVector<CalleeSavedInfo, 8>
+  getUnmanagedCSI(const MachineFunction &MF,
+                  const std::vector<CalleeSavedInfo> &CSI,
+                  bool ReverseOrder = false) const;
+
   StackOffset getFrameIndexReference(const MachineFunction &MF, int FI,
                                      Register &FrameReg) const override;
 
@@ -93,6 +98,8 @@ class RISCVFrameLowering : public TargetFrameLowering {
 
   uint64_t getStackThreshold() const override;
 
+  bool enableCSRSaveRestorePointsSplit() const override;
+
 protected:
   const RISCVSubtarget &STI;
 
diff --git a/llvm/lib/Target/RISCV/RISCVRegisterInfo.cpp b/llvm/lib/Target/RISCV/RISCVRegisterInfo.cpp
index adc6eadee2e351..91c1fc949e8272 100644
--- a/llvm/lib/Target/RISCV/RISCVRegisterInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVRegisterInfo.cpp
@@ -595,10 +595,33 @@ static unsigned getXqciloWideOpcode(unsigned Opc) {
   }
 }
 
+int64_t RISCVRegisterInfo::getCSIFrameOffset(MachineFunction *MF) const {
+  uint64_t FirstSPAdjustAmount =
+      getFrameLowering(*MF)->getFirstSPAdjustAmount(*MF);
+  if (FirstSPAdjustAmount)
+    return getFrameLowering(*MF)->getStackSizeWithRVVPadding(*MF) -
+           FirstSPAdjustAmount;
+  return 0;
+}
+
+bool RISCVRegisterInfo::isCSIFrameIndex(MachineFunction *MF,
+                                        int FrameIndex) const {
+  const MachineFrameInfo &MFI = MF->getFrameInfo();
+  const auto &CSI =
+      getFrameLowering(*MF)->getUnmanagedCSI(*MF, MFI.getCalleeSavedInfo());
+  if (!CSI.empty()) {
+    int MinCSFI = CSI.front().getFrameIdx();
+    int MaxCSFI = CSI.back().getFrameIdx();
+    if (FrameIndex >= MinCSFI && FrameIndex <= MaxCSFI)
+      return true;
+  }
+  return false;
+}
+
 bool RISCVRegisterInfo::eliminateFrameIndex(MachineBasicBlock::iterator II,
                                             int SPAdj, unsigned FIOperandNum,
                                             RegScavenger *RS) const {
-  assert(SPAdj == 0 && "Unexpected non-zero SPAdj value");
+  // assert(SPAdj == 0 && "Unexpected non-zero SPAdj value");
 
   MachineInstr &MI = *II;
   MachineFunction &MF = *MI.getParent()->getParent();
@@ -606,12 +629,16 @@ bool RISCVRegisterInfo::eliminateFrameIndex(MachineBasicBlock::iterator II,
   const RISCVSubtarget &ST = MF.getSubtarget<RISCVSubtarget>();
   const RISCVInstrInfo *TII = ST.getInstrInfo();
   bool Is64Bit = ST.is64Bit();
+  const MachineFrameInfo &MFI = MF.getFrameInfo();
   DebugLoc DL = MI.getDebugLoc();
 
   int FrameIndex = MI.getOperand(FIOperandNum).getIndex();
   Register FrameReg;
   StackOffset Offset =
       getFrameLowering(MF)->getFrameIndexReference(MF, FrameIndex, FrameReg);
+
+  Offset += StackOffset::getFixed(SPAdj);
+
   bool IsRVVSpill = RISCV::isRVVSpill(MI);
   if (!IsRVVSpill)
     Offset += StackOffset::getFixed(MI.getOperand(FIOperandNum + 1).getImm());
diff --git a/llvm/lib/Target/RISCV/RISCVRegisterInfo.h b/llvm/lib/Target/RISCV/RISCVRegisterInfo.h
index e94f5f0b4b364c..f50558f95e88fd 100644
--- a/llvm/lib/Target/RISCV/RISCVRegisterInfo.h
+++ b/llvm/lib/Target/RISCV/RISCVRegisterInfo.h
@@ -104,6 +104,9 @@ struct RISCVRegisterInfo : public RISCVGenRegisterInfo {
   bool eliminateFrameIndex(MachineBasicBlock::iterator MI, int SPAdj,
                            unsigned FIOperandNum,
                            RegScavenger *RS = nullptr) const override;
+  bool isCSIFrameIndex(MachineFunction *MF, int FrameIndex) const override;
+
+  int64_t getCSIFrameOffset(MachineFunction *MF) const override;
 
   bool requiresVirtualBaseRegisters(const MachineFunction &MF) const override;
 
diff --git a/llvm/test/CodeGen/RISCV/shrinkwrap-split.mir b/llvm/test/CodeGen/RISCV/shrinkwrap-split.mir
new file mode 100644
index 00000000000000..b6c89e9de2e3ed
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/shrinkwrap-split.mir
@@ -0,0 +1,284 @@
+# RUN: llc -march=riscv64 -run-pass shrink-wrapping  %s -o - | FileCheck  %s
+
+# CHECK:       savePoint:
+# CHECK-NEXT:    - point:           '%bb.0'
+# CHECK-NEXT:      registers:
+# CHECK-NEXT:        - '$x9'
+# CHECK-NEXT:    - point:           '%bb.9'
+# CHECK-NEXT:      registers:
+# CHECK-NEXT:        - '$x18'
+# CHECK-NEXT:        - '$x19'
+# CHECK-NEXT:        - '$x20'
+# CHECK-NEXT:        - '$x21'
+# CHECK-NEXT:        - '$x22'
+# CHECK-NEXT:        - '$x23'
+# CHECK-NEXT:        - '$x24'
+# CHECK-NEXT:        - '$x25'
+# CHECK-NEXT:        - '$x8'
+# CHECK-NEXT:  restorePoint:
+# CHECK-NEXT:    - point:           '%bb.10'
+# CHECK-NEXT:      registers:
+# CHECK-NEXT:        - '$x18'
+# CHECK-NEXT:        - '$x19'
+# CHECK-NEXT:        - '$x20'
+# CHECK-NEXT:        - '$x21'
+# CHECK-NEXT:        - '$x22'
+# CHECK-NEXT:        - '$x23'
+# CHECK-NEXT:        - '$x24'
+# CHECK-NEXT:        - '$x25'
+# CHECK-NEXT:        - '$x8'
+# CHECK-NEXT:    - point:           '%bb.8'
+# CHECK-NEXT:      registers:
+# CHECK-NEXT:        - '$x9'
+
+--- |
+  ; ModuleID = 'shrinkwrap-split.ll'
+  %struct.task = type { i32, i32, [20 x i32] }
+
+  define i32 @test(ptr %t, i32 %i, i1 %cond) {
+  entry:
+    %arr = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2
+    %0 = load i32, ptr %arr, align 4
+    %arrayidx2 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 1
+    %1 = load i32, ptr %arrayidx2, align 4
+    %arrayidx4 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 2
+    %2 = load i32, ptr %arrayidx4, align 4
+    %arrayidx6 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 3
+    %3 = load i32, ptr %arrayidx6, align 4
+    %arrayidx8 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 4
+    %4 = load i32, ptr %arrayidx8, align 4
+    %arrayidx10 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 5
+    %5 = load i32, ptr %arrayidx10, align 4
+    %arrayidx12 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 6
+    %6 = load i32, ptr %arrayidx12, align 4
+    %arrayidx14 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 7
+    %7 = load i32, ptr %arrayidx14, align 4
+    %arrayidx16 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 8
+    %8 = load i32, ptr %arrayidx16, align 4
+    %arrayidx18 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 9
+    %9 = load i32, ptr %arrayidx18, align 4
+    %arrayidx20 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 10
+    %10 = load i32, ptr %arrayidx20, align 4
+    %arrayidx22 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 11
+    %11 = load i32, ptr %arrayidx22, align 4
+    %arrayidx24 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 12
+    %12 = load i32, ptr %arrayidx24, align 4
+    %arrayidx26 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 13
+    %13 = load i32, ptr %arrayidx26, align 4
+    %arrayidx28 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 14
+    %14 = load i32, ptr %arrayidx28, align 4
+    %arrayidx30 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 15
+    %15 = load i32, ptr %arrayidx30, align 4
+    %arrayidx32 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 16
+    %16 = load i32, ptr %arrayidx32, align 4
+    %arrayidx34 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 17
+    %17 = load i32, ptr %arrayidx34, align 4
+    %arrayidx36 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 18
+    %18 = load i32, ptr %arrayidx36, align 4
+    %arrayidx38 = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 2, i64 19
+    %19 = load i32, ptr %arrayidx38, align 4
+    %20 = load i32, ptr %t, align 4
+    %add = add i32 %10, %0
+    %add39 = add i32 %add, %20
+    %cmp = icmp slt i32 %add39, %i
+    br i1 %cmp, label %for.cond.preheader, label %cleanup
+
+  for.cond.preheader:                               ; preds = %entry
+    %y = getelementptr inbounds %struct.task, ptr %t, i64 0, i32 1
+    %21 = load i32, ptr %y, align 4
+    %cmp40.not119 = icmp eq i32 %21, 0
+    br i1 %cmp40.not119, label %for.cond.cleanup, label %for.body.preheader
+
+  for.body.preheader:                               ; preds = %for.cond.preheader
+    %22 = add i32 %21, -1
+    %cond41 = select i1 %cond, i32 %22, i32 %i
+    store i32 %cond41, ptr %t, align 4
+    br label %for.cond.cleanup
+
+  for.cond.cleanup:                                 ; preds = %for.body.preheader, %for.cond.preheader
+    %23 = phi i32 [ %cond41, %for.body.preheader ], [ %20, %for.cond.preheader ]
+    %tobool44 = icmp ne i32 %21, 0
+    %conv = zext i1 %tobool44 to i32
+    %add48 = add i32 %1, %0
+    %add49 = add i32 %add48, %2
+    %add50 = add i32 %add49, %3
+    %add51 = add i32 %add50, %4
+    %add52 = add i32 %add51, %5
+    %add53 = add i32 %add52, %6
+    %add54 = add i32 %add53, %7
+    %add55 = add i32 %add54, %8
+    %add56 = add i32 %add55, %9
+    %add57 = add i32 %add56, %10
+    %add58 = add i32 %add57, %11
+    %add59 = add i32 %add58, %12
+    %add60 = add i32 %add59, %13
+    %add61 = add i32 %add60, %14
+    %add62 = add i32 %add61, %15
+    %add63 = add i32 %add62, %16
+    %add64 = add i32 %add63, %17
+    %add65 = add i32 %add64, %18
+    %add66 = add i32 %add65, %19
+    %add67 = add i32 %add66, %conv
+    %add68 = add i32 %add67, %23
+    br label %cleanup
+
+  cleanup:                                          ; preds = %for.cond.cleanup, %entry
+    %retval.0 = phi i32 [ %add68, %for.cond.cleanup ], [ %i, %entry ]
+    ret i32 %retval.0
+  }
+...
+---
+name: test
+alignment:       2
+exposesReturnsTwice: false
+legalized:       false
+regBankSelected: false
+selected:        false
+failedISel:      false
+tracksRegLiveness: true
+hasWinCFI:       false
+callsEHReturn:   false
+callsUnwindInit: false
+hasEHScopes:     false
+hasEHFunclets:   false
+isOutlined:      false
+debugInstrRef:   false
+failsVerification: false
+tracksDebugUserValues: true
+registers:       []
+liveins:
+  - { reg: '$x10', virtual-reg: '' }
+  - { reg: '$x11', virtual-reg: '' }
+  - { reg: '$x12', virtual-reg: '' }
+frameInfo:
+  isFrameAddressTaken: false
+  isReturnAddressTaken: false
+  hasStackMap:     false
+  hasPatchPoint:   false
+  stackSize:       0
+  offsetAdjustment: 0
+  maxAlignment:    1
+  adjustsStack:    false
+  hasCalls:        false
+  stackProtector:  ''
+  functionContext: ''
+  maxCallFrameSize: 4294967295
+  cvBytesOfCalleeSavedRegisters: 0
+  hasOpaqueSPAdjustment: false
+  hasVAStart:      false
+  hasMustTailInVarArgFunc: false
+  hasTailCall:     false
+  localFrameSize:  0
+fixedStack:      []
+stack:           []
+entry_values:    []
+callSites:       []
+debugValueSubstitutions: []
+constants:       []
+machineFunctionInfo:
+  varArgsFrameIndex: 0
+  varArgsSaveSize: 0
+body:             |
+  bb.0.entry:
+    successors: %bb.1(0x40000000), %bb.8(0x40000000)
+    liveins: $x10, $x11, $x12
+
+    renamable $x13 = COPY $x10
+    renamable $x9 = LW $x10, 8 :: (load (s32) from %ir.arr)
+    renamable $x16 = LW $x10, 48 :: (load (s32) from %ir.arrayidx20)
+    renamable $x15 = LW $x10, 0 :: (load (s32) from %ir.t)
+    renamable $x14 = ADD renamable $x16, renamable $x9
+    renamable $x10 = COPY $x11
+    renamable $x14 = ADDW killed renamable $x14, renamable $x15
+    BLT killed renamable $x14, $x11, %bb.1
+
+  bb.8:
+    successors: %bb.7(0x80000000)
+    liveins: $x10
+
+    PseudoBR %bb.7
+
+  bb.1.for.cond.preheader:
+    successors: %bb.6(0x30000000), %bb.2(0x50000000)
+    liveins: $x9, $x10, $x12, $x13, $x15, $x16
+
+    renamable $x20 = LW renamable $x13, 12 :: (load (s32) from %ir.arrayidx2)
+    renamable $x30 = LW renamable $x13, 16 :: (load (s32) from %ir.arrayidx4)
+    renamable $x24 = LW renamable $x13, 20 :: (load (s32) from %ir.arrayidx6)
+    renamable $x31 = LW renamable $x13, 24 :: (load (s32) from %ir.arrayidx8)
+    renamable $x25 = LW renamable $x13, 28 :: (load (s32) from %ir.arrayidx10)
+    renamable $x7 = LW renamable $x13, 32 :: (load (s32) from %ir.arrayidx12)
+    renamable $x21 = LW renamable $x13, 36 :: (load (s32) from %ir.arrayidx14)
+    renamable $x14 = LW renamable $x13, 40 :: (load (s32) from %ir.arrayidx16)
+    renamable $x23 = LW renamable $x13, 44 :: (load (s32) from %ir.arrayidx18)
+    renamable $x29 = LW renamable $x13, 52 :: (load (s32) from %ir.arrayidx22)
+    renamable $x22 = LW renamable $x13, 56 :: (load (s32) from %ir.arrayidx24)
+    renamable $x6 = LW renamable $x13, 60 :: (load (s32) from %ir.arrayidx26)
+    renamable $x17 = LW renamable $x13, 64 :: (load (s32) from %ir.arrayidx28)
+    renamable $x5 = LW renamable $x13, 68 :: (load (s32) from %ir.arrayidx30)
+    renamable $x28 = LW renamable $x13, 72 :: (load (s32) from %ir.arrayidx32)
+    renamable $x18 = LW renamable $x13, 76 :: (load (s32) from %ir.arrayidx34)
+    renamable $x19 = LW renamable $x13, 80 :: (load (s32) from %ir.arrayidx36)
+    renamable $x8 = LW renamable $x13, 4 :: (load (s32) from %ir.y)
+    renamable $x11 = LW renamable $x13, 84 :: (load (s32) from %ir.arrayidx38)
+    BEQ renamable $x8, $x0, %bb.6
+    PseudoBR %bb.2
+
+  bb.2.for.body.preheader:
+    successors: %bb.4(0x40000000), %bb.3(0x40000000)
+    liveins: $x5, $x6, $x7, $x8, $x9, $x10, $x11, $x12, $x13, $x14, $x16, $x17, $x18, $x19, $x20, $x21, $x22, $x23, $x24, $x25, $x28, $x29, $x30, $x31
+
+    BEQ killed renamable $x12, $x0, %bb.4
+
+  bb.3:
+    successors: %bb.5(0x80000000)
+    liveins: $x5, $x6, $x7, $x8, $x9, $x11, $x13, $x14, $x16, $x17, $x18, $x19, $x20, $x21, $x22, $x23, $x24, $x25, $x28, $x29, $x30, $x31
+
+    renamable $x10 = ADDIW renamable $x8, -1
+    PseudoBR %bb.5
+
+  bb.4.for.body.preheader:
+    successors: %bb.5(0x80000000)
+    liveins: $x5, $x6, $x7, $x8, $x9, $x10, $x11, $x13, $x14, $x16, $x17, $x18, $x19, $x20, $x21, $x22, $x23, $x24, $x25, $x28, $x29, $x30, $x31
+
+
+  bb.5.for.body.preheader:
+    successors: %bb.6(0x80000000)
+    liveins: $x5, $x6, $x7, $x8, $x9, $x10, $x11, $x13, $x14, $x16, $x17, $x18, $x19, $x20, $x21, $x22, $x23, $x24, $x25, $x28, $x29, $x30, $x31
+
+    SW renamable $x10, killed renamable $x13, 0 :: (store (s32) into %ir.t)
+    renamable $x15 = COPY killed renamable $x10
+
+  bb.6.for.cond.cleanup:
+    successors: %bb.7(0x80000000)
+    liveins: $x5, $x6, $x7, $x8, $x9, $x11, $x14, $x15, $x16, $x17, $x18, $x19, $x20, $x21, $x22, $x23, $x24, $x25, $x28, $x29, $x30, $x31
+
+    renamable $x9 = ADD killed renamable $x20, killed renamable $x9
+    renamable $x30 = ADD killed renamable $x30, killed renamable $x24
+    renamable $x30 = ADD killed renamable $x9, killed renamable $x30
+    renamable $x31 = ADD killed renamable $x31, killed renamable $x25
+    renamable $x14 = ADD killed renamable $x21, killed renamable $x14
+    renamable $x7 = ADD killed renamable $x31, killed renamable $x7
+    renamable $x14 = ADD killed renamable $x14, killed renamable $x23
+    renamable $x7 = ADD killed renamable $x30, killed renamable $x7
+    renamable $x14 = ADD killed renamable $x14, killed renamable $x16
+    renamable $x10 = SLTU $x0, killed renamable $x8
+    renamable $x14 = ADD killed renamable $x7, killed renamable $x14
+    renamable $x29 = ADD killed renamable $x29, killed renamable $x22
+    renamable $x28 = ADD killed renamable $x28, killed renamable $x18
+    renamable $x6 = ADD killed renamable $x29, killed renamable $x6
+    renamable $x28 = ADD killed renamable $x28, killed renamable $x19
+    renamable $x17 = ADD killed renamable $x6, killed renamable $x17
+    renamable $x11 = ADD killed renamable $x28, killed renamable $x11
+    renamable $x17 = ADD killed renamable $x17, killed renamable $x5
+    renamable $x10 = ADD killed renamable $x11, killed renamable $x10
+    renamable $x14 = ADD killed renamable $x14, killed renamable $x17
+    renamable $x10 = ADD killed renamable $x10, killed renamable $x15
+    renamable $x10 = ADDW killed renamable $x14, killed renamable $x10
+
+  bb.7.cleanup:
+    liveins: $x10
+
+    PseudoRET implicit $x10
+
+...



More information about the llvm-commits mailing list