[llvm] AMDGPU: Add NextUseAnalysis Pass (PR #178873)

Matt Arsenault via llvm-commits llvm-commits at lists.llvm.org
Fri Apr 10 10:15:06 PDT 2026


================
@@ -0,0 +1,2614 @@
+//===---------------------- AMDGPUNextUseAnalysis.cpp ---------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// This file implements the AMDGPUNextUseAnalysis pass, a machine-level analysis
+// that computes the distance from each instruction to the "nearest" next use of
+// every live virtual register. These distances guide register spilling
+// decisions by identifying which live values are furthest from their next use
+// and are therefore the best candidates to spill.
+//
+// The analysis is based on the Braun & Hack CC'09 paper "Register Spilling and
+// Live-Range Splitting for SSA-Form Programs."
+//
+// Key concepts:
+//
+//   NextUseDistance     A loop-depth-weighted instruction count representing
+//                       how far away a register's next use is. Distances
+//                       through deeper loops are scaled by fromLoopDepth() so
+//                       that uses inside hot loops appear closer.
+//
+//   Inter-block         Pre-computed shortest weighted distances between all
+//   distances           pairs of basic blocks, used to efficiently answer
+//                       cross-block next-use queries. Each intermediate block
+//                       is weighted by fromLoopDepth() applied once per loop
+//                       boundary crossing relative to the destination.
+//
+// Configuration flags (see Config struct in the header):
+//
+//   CountPhis           Count PHI instructions toward distance and block size.
+//   ForwardOnly         Restrict inter-block distances to forward-reachable
+//                       paths.
+//   PreciseUseModeling  Model PHI uses at their incoming edge block and filter
+//                       uses with intermediate redefinitions.
+//   PromoteToPreheader  Route loop-entry and inner-loop uses to the preheader.
+//
+// This file contains:
+//
+//   - Command-line options for configuration and debug output
+//   - LiveRegUse / JSON helpers
+//   - AMDGPUNextUseAnalysisImpl (the main analysis implementation)
+//       - Instruction ID assignment and block size computation
+//       - CFG path pre-computation (reachability, loop depth, back-edges)
+//       - Inter-block distance computation
+//       - Per-register next-use distance queries and caching
+//   - AMDGPUNextUseAnalysis (public facade, pimpl)
+//   - Legacy and new pass manager wrappers
+//
+//===----------------------------------------------------------------------===//
+
+#include "AMDGPUNextUseAnalysis.h"
+#include "AMDGPU.h"
+#include "GCNRegPressure.h"
+#include "GCNSubtarget.h"
+
+#include "llvm/ADT/DenseMap.h"
+#include "llvm/ADT/PostOrderIterator.h"
+#include "llvm/ADT/SmallVector.h"
+#include "llvm/CodeGen/MachineBasicBlock.h"
+#include "llvm/CodeGen/MachineFunction.h"
+#include "llvm/CodeGen/MachineInstr.h"
+#include "llvm/CodeGen/MachineLoopInfo.h"
+#include "llvm/IR/ModuleSlotTracker.h"
+#include "llvm/InitializePasses.h"
+#include "llvm/Support/FileSystem.h"
+#include "llvm/Support/JSON.h"
+#include "llvm/Support/Timer.h"
+#include "llvm/Support/ToolOutputFile.h"
+#include "llvm/Support/raw_ostream.h"
+
+#include <algorithm>
+#include <limits>
+#include <string>
+
+using namespace llvm;
+
+#define DEBUG_TYPE "amdgpu-next-use-analysis"
+
+//==============================================================================
+// Options etc
+//==============================================================================
+namespace {
+
+cl::opt<bool> DumpNextUseDistance("amdgpu-next-use-analysis-dump-distance",
+                                  cl::init(false), cl::Hidden);
+
+cl::opt<bool>
+    DistanceCacheEnabled("amdgpu-next-use-analysis-distance-cache",
+                         cl::init(true), cl::Hidden,
+                         cl::desc("Enable live-reg-use distance cache"));
+
+cl::opt<std::string>
+    DumpNextUseDistanceAsJson("amdgpu-next-use-analysis-dump-distance-as-json",
+                              cl::Hidden);
+cl::opt<bool>
+    DumpNextUseDistanceVerbose("amdgpu-next-use-analysis-dump-distance-verbose",
+                               cl::init(false), cl::Hidden);
+
+// 'graphics' and 'compute' modes arose due to initial competing implementations
+// of next-use analysis that emphasized different types of workloads. This
+// implementation is a compromise that combines aspects of both. Over time, the
+// hope is we will be able to remove some of these differences and settle on a
+// more unified implementation.
+cl::opt<std::string>
+    ConfigPresetOpt("amdgpu-next-use-analysis-config", cl::Hidden,
+                    cl::init("graphics"),
+                    cl::desc("Config preset: 'graphics' or 'compute'"));
+
+cl::opt<bool> ConfigCountPhisOpt(
+    "amdgpu-next-use-analysis-count-phis", cl::Hidden,
+    cl::desc("Count PHI instructions toward distance and block size"));
+cl::opt<bool> ConfigForwardOnlyOpt(
+    "amdgpu-next-use-analysis-forward-only", cl::Hidden,
+    cl::desc("Restrict inter-block distances to forward-reachable paths"));
+cl::opt<bool> ConfigPreciseUseModelingOpt(
+    "amdgpu-next-use-analysis-precise-use-modeling", cl::Hidden,
+    cl::desc("Model PHI uses via incoming edge block with loop-aware "
+             "reachability filtering"));
+cl::opt<bool> ConfigPromoteToPreheaderOpt(
+    "amdgpu-next-use-analysis-use-preheader-model", cl::Hidden,
+    cl::desc("Promote loop-entry and inner-loop uses to the loop preheader"));
+} // namespace
+
+//==============================================================================
+// LiveRegUse - Represents a live register use with its distance. Used for
+// tracking and sorting register uses by distance.
+//==============================================================================
+namespace {
+using UseDistancePair = AMDGPUNextUseAnalysis::UseDistancePair;
+struct LiveRegUse : public UseDistancePair {
+  // 'nullptr' indicates an unset/invalid state.
+  LiveRegUse() : UseDistancePair(nullptr, 0) {}
+  LiveRegUse(const MachineOperand *Use, NextUseDistance Dist)
+      : UseDistancePair(Use, Dist) {}
+  LiveRegUse(const UseDistancePair &P) : UseDistancePair(P) {}
+
+  bool isUnset() const { return Use == nullptr; }
+
+  Register getReg() const { return Use->getReg(); }
+  unsigned getSubReg() const { return Use->getSubReg(); }
+  LaneBitmask getLaneMask(const SIRegisterInfo *TRI) const {
+    return TRI->getSubRegIndexLaneMask(Use->getSubReg());
+  }
+
+  bool isCloserThan(const LiveRegUse &X) const {
+    if (Dist < X.Dist)
+      return true;
+
+    if (Dist > X.Dist)
+      return false;
+
+    if (Use == X.Use)
+      return false;
+
+    // Ugh. When !CountPhis, PHIs and the first non-PHI instruction have id
+    // 0. In this case, consider PHIs as less than the first non-PHI
+    // instruction.
+    const MachineInstr *ThisMI = Use->getParent();
+    const MachineInstr *XMI = X.Use->getParent();
+    const MachineBasicBlock *ThisMBB = ThisMI->getParent();
+    if (ThisMBB == XMI->getParent()) {
+      if (ThisMI->isPHI() && !XMI->isPHI() &&
+          XMI == &(*ThisMBB->getFirstNonPHI()))
+        return true;
+    }
+
+    // Ensure deterministic results
+    return X.getReg() < getReg();
+  }
+
+  void print(raw_ostream &OS) const {
+    if (isUnset()) {
+      OS << "<unset>";
+      return;
+    }
+    Dist.print(OS);
+    OS << " [" << printReg(getReg());
+    if (getSubReg())
+      OS << ":" << getSubReg();
+    OS << "]";
+  }
+
+  LLVM_DUMP_METHOD void dump() const {
+    print(dbgs());
+    dbgs() << '\n';
+  }
+};
+
+inline bool updateClosest(LiveRegUse &Closest, const LiveRegUse &X) {
+  if (!Closest.Use || X.isCloserThan(Closest)) {
+    Closest = X;
+    return true;
+  }
+  return false;
+}
+
+inline bool updateFurthest(LiveRegUse &Furthest, const LiveRegUse &X) {
+  if (!Furthest.Use || Furthest.isCloserThan(X)) {
+    Furthest = X;
+    return true;
+  }
+  return false;
+}
+} // namespace
+
+//==============================================================================
+// JSON helpers
+//==============================================================================
+namespace {
+template <typename Lambda>
+void printStringAttr(json::OStream &J, const char *Name, Lambda L) {
+  J.attributeBegin(Name);
+  raw_ostream &OS = J.rawValueBegin();
+  OS << '"';
+  L(OS);
+  OS << '"';
+  J.rawValueEnd();
+  J.attributeEnd();
+}
+void printStringAttr(json::OStream &J, const char *Name, Printable P) {
+  printStringAttr(J, Name, [&](raw_ostream &OS) { OS << P; });
+}
+
+void printStringAttr(json::OStream &J, const char *Name, const MachineInstr &MI,
+                     ModuleSlotTracker &MST) {
+  printStringAttr(J, Name, [&](raw_ostream &OS) {
+    MI.print(OS, MST,
+             /* IsStandalone    */ false,
+             /* SkipOpers       */ false,
+             /* SkipDebugLoc    */ false,
+             /* AddNewLine ---> */ false,
+             /* TargetInstrInfo */ nullptr);
+  });
+}
+
+void printMBBNameAttr(json::OStream &J, const char *Name,
+                      const MachineBasicBlock &MBB, ModuleSlotTracker &MST) {
+  printStringAttr(J, Name, [&](raw_ostream &OS) {
+    MBB.printName(OS, MachineBasicBlock::PrintNameIr, &MST);
+  });
+}
+
+template <typename NameLambda, typename ValueT>
+void printAttr(json::OStream &J, NameLambda NL, ValueT V) {
+  std::string Name;
+  raw_string_ostream NameOS(Name);
+  NL(NameOS);
+  J.attribute(NameOS.str(), V);
+}
+
+template <typename ValueT>
+void printAttr(json::OStream &J, const Printable &P, ValueT V) {
+  printAttr(J, [&](raw_ostream &OS) { OS << P; }, V);
+}
+
+} // namespace
+
+//==============================================================================
+// AMDGPUNextUseAnalysisImpl
+//==============================================================================
+class llvm::AMDGPUNextUseAnalysisImpl {
+public:
+  struct CacheableNextUseDistance {
+    bool IsInstrRelative;
+    NextUseDistance Distance;
+  };
+  static constexpr bool InstrRelative = true;
+  static constexpr bool InstrInvariant = false;
+
+private:
+  const MachineFunction *MF = nullptr;
+  const SIRegisterInfo *TRI = nullptr;
+  const SIInstrInfo *TII = nullptr;
+  const MachineLoopInfo *MLI = nullptr;
+  const MachineRegisterInfo *MRI = nullptr;
+
+  using InstrIdTy = unsigned;
+  using InstrToIdMap = DenseMap<const MachineInstr *, InstrIdTy>;
+  InstrToIdMap InstrToId;
+  AMDGPUNextUseAnalysis::Config Cfg;
+
+  void initializeTables() {
+    for (const MachineBasicBlock &BB : *MF)
+      calcInstrIds(&BB, InstrToId);
+    initializeCfgPaths();
+    initializeInterBlockDistances();
+  }
+
+  void clearTables() {
+    InstrToId.clear();
+    RegUseMap.clear();
+    Paths.clear();
+
+    resetDistanceCache();
+  }
+
+  //~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+  // Instruction Ids
+  //~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+private:
+  unsigned sizeOf(const MachineInstr &MI) const {
+    // When !Cfg.CountPhis, PHIs do not contribute to distances/sizes since they
+    // generally don't result in the generation of a machine instruction.
+    // FIXME: Consider using MI.isPseudo()
+    return Cfg.CountPhis ? 1 : !MI.isPHI();
+  }
+
+  void calcInstrIds(const MachineBasicBlock *BB,
+                    InstrToIdMap &MutableInstrToId) const {
+    InstrIdTy Id = 0;
+    for (auto &MI : BB->instrs()) {
+      MutableInstrToId[&MI] = Id;
+      Id += sizeOf(MI);
+    }
+  }
+
+  /// Returns MI's instruction Id. It renumbers (part of) the BB if MI is not
+  /// found in the map.
+  InstrIdTy getInstrId(const MachineInstr *MI) const {
+    auto It = InstrToId.find(MI);
+    if (It != InstrToId.end())
+      return It->second;
+
+    // Renumber the MBB.
+    // TODO: Renumber from MI onwards.
+    auto &MutableInstrToId = const_cast<InstrToIdMap &>(InstrToId);
+    calcInstrIds(MI->getParent(), MutableInstrToId);
+    return InstrToId.find(MI)->second;
+  }
+
+  // Length of the segment from MI (inclusive) to the first instruction of the
+  // basic block.
+  InstrIdTy getHeadLen(const MachineInstr *MI) const {
+    const MachineBasicBlock *MBB = MI->getParent();
+    return getInstrId(MI) + getInstrId(&MBB->instr_front()) + 1;
+  }
+
+  // Length of the segment from MI (exclusive) to the last instruction of the
+  // basic block.
+  InstrIdTy getTailLen(const MachineInstr *MI) const {
+    const MachineBasicBlock *MBB = MI->getParent();
+    return getInstrId(&MBB->instr_back()) - getInstrId(MI);
+  }
+
+  // Length of the segment from 'From' to 'To' (exclusive). Both instructions
+  // must be in the same basic block.
+  InstrIdTy getDistance(const MachineInstr *From,
+                        const MachineInstr *To) const {
+    assert(From->getParent() == To->getParent());
+    return getInstrId(To) - getInstrId(From);
+  }
+
+  //~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+  // RegUses - cache of uses by register
+  //~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+private:
+  DenseMap<Register, SmallVector<const MachineOperand *>> RegUseMap;
+
+  const SmallVector<const MachineOperand *> &
+  getRegisterUses(Register Reg) const {
+    auto I = RegUseMap.find(Reg);
+    if (I != RegUseMap.end())
+      return I->second;
+
+    auto *NonConstThis = const_cast<AMDGPUNextUseAnalysisImpl *>(this);
+    SmallVector<const MachineOperand *> &Uses = NonConstThis->RegUseMap[Reg];
+    for (const MachineOperand &UseMO : MRI->use_nodbg_operands(Reg)) {
+      if (!UseMO.isUndef())
+        Uses.push_back(&UseMO);
+    }
+    return Uses;
+  }
+
+  bool hasAtLeastOneUse(Register Reg) const {
+    return !getRegisterUses(Reg).empty();
+  }
+
+  //~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+  // Paths
+  //~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+private:
+  class Path
+      : public std::pair<const MachineBasicBlock *, const MachineBasicBlock *> {
+  public:
+    using Base =
+        std::pair<const MachineBasicBlock *, const MachineBasicBlock *>;
+    using Base::pair;
+    Path(const Base &Pair) : Base(Pair) {}
+
+    const MachineBasicBlock *src() const { return first; }
+    const MachineBasicBlock *dst() const { return second; }
+
+    using DenseMapInfo = llvm::DenseMapInfo<Base>;
+  };
+
+  enum EdgeKind { Back = -1, None = 0, Forward = 1 };
+  struct PathInfo {
+    EdgeKind EK;
+    bool Reachable;
+    int ForwardReachable;
+    unsigned RelativeLoopDepth;
+    std::optional<NextUseDistance> ShortestDistance;
+    std::optional<NextUseDistance> ShortestUnweightedDistance;
+    InstrIdTy Size;
+
+    PathInfo()
+        : EK(None), Reachable(false), ForwardReachable(-1),
+          RelativeLoopDepth(0), Size(0) {}
+
+    bool isBackedge() const { return EK == EdgeKind::Back; }
+
+    bool isForwardReachableSet() const { return 0 <= ForwardReachable; }
+    bool isForwardReachableUnset() const { return ForwardReachable < 0; }
+    bool isForwardReachable() const { return ForwardReachable == 1; }
+    bool isNotForwardReachable() const { return ForwardReachable == 0; }
+
+    void print(raw_ostream &OS) const {
+      const char *EKStr = EK == Back ? "back" : EK == Forward ? "fwd" : "none";
+      OS << "{ek=" << EKStr << " reach=" << Reachable
+         << " fwd-reach=" << ForwardReachable
+         << " loop-depth=" << RelativeLoopDepth << " size=" << Size;
+      if (ShortestDistance) {
+        OS << " shdist=";
+        ShortestDistance->print(OS);
+      }
+      if (ShortestUnweightedDistance) {
+        OS << " shudist=";
+        ShortestUnweightedDistance->print(OS);
+      }
+      OS << "}";
+    }
+
+    LLVM_DUMP_METHOD void dump() const {
+      print(dbgs());
+      dbgs() << '\n';
+    }
+  };
+
+  //----------------------------------------------------------------------------
+  // Path Storage - 'Paths' is lazily populated and some members are lazily
+  // computed. All mutations should go through one of the 'initializePathInfo*'
+  // flavors below.
+  //----------------------------------------------------------------------------
+  DenseMap<Path, PathInfo, Path::DenseMapInfo> Paths;
+
+  const PathInfo *maybePathInfoFor(const MachineBasicBlock *From,
+                                   const MachineBasicBlock *To) const {
+    auto I = Paths.find({From, To});
+    return I == Paths.end() ? nullptr : &I->second;
+  }
+
+  PathInfo &getOrInitPathInfo(const MachineBasicBlock *From,
+                              const MachineBasicBlock *To) const {
+    auto *NonConstThis = const_cast<AMDGPUNextUseAnalysisImpl *>(this);
+    auto &MutablePaths = NonConstThis->Paths;
+
+    Path P(From, To);
+    auto [I, Inserted] = MutablePaths.try_emplace(P);
+    if (!Inserted)
+      return I->second;
+
+    bool Reachable = calcIsReachable(P.src(), P.dst());
+
+    // Iterator may have been invalidated by calcIsReachable, so get a fresh
+    // reference to the slot.
+    return NonConstThis->initializePathInfo(MutablePaths.at(P), P,
+                                            EdgeKind::None, Reachable);
+  }
+
+  const PathInfo &pathInfoFor(const MachineBasicBlock *From,
+                              const MachineBasicBlock *To) const {
+    return getOrInitPathInfo(From, To);
+  }
+
+  //----------------------------------------------------------------------------
+  // initializePathInfo* - various flavors of PathInfo initialization. They
+  // (should) always funnel to the first flavor below.
+  //----------------------------------------------------------------------------
+  PathInfo &initializePathInfo(PathInfo &Slot, Path P, EdgeKind EK,
+                               bool Reachable) {
+    Slot.EK = EK;
+    Slot.Reachable = Reachable;
+    Slot.ForwardReachable = EK != EdgeKind::None ? (0 < EK) : -1;
+    Slot.RelativeLoopDepth =
+        Slot.Reachable ? calcRelativeLoopDepth(P.src(), P.dst()) : 0;
+    Slot.Size = P.src() == P.dst() ? calcSize(P.src()) : 0;
+    if (EK != EdgeKind::None)
+      Slot.ShortestUnweightedDistance = 0;
+    return Slot;
+  }
+
+  PathInfo &initializePathInfo(Path P, EdgeKind EK, bool Reachable) const {
+    auto *NonConstThis = const_cast<AMDGPUNextUseAnalysisImpl *>(this);
+    auto &MutablePaths = NonConstThis->Paths;
+    return NonConstThis->initializePathInfo(MutablePaths[P], P, EK, Reachable);
+  }
+
+  std::pair<PathInfo *, bool> maybeInitializePathInfo(Path P, EdgeKind EK,
+                                                      bool Reachable) const {
+    auto *NonConstThis = const_cast<AMDGPUNextUseAnalysisImpl *>(this);
+    auto &MutablePaths = NonConstThis->Paths;
+    auto [I, Inserted] = MutablePaths.try_emplace(P);
+    if (Inserted)
+      NonConstThis->initializePathInfo(I->second, P, EK, Reachable);
+    return {&I->second, Inserted};
+  }
+
+  bool initializePathInfoForwardReachable(const MachineBasicBlock *From,
+                                          const MachineBasicBlock *To,
+                                          bool Value) const {
+    PathInfo &Slot = getOrInitPathInfo(From, To);
+    assert(Slot.isForwardReachableUnset());
+    Slot.ForwardReachable = Value;
+    return Value;
+  }
+
+  NextUseDistance
+  initializePathInfoShortestDistance(const MachineBasicBlock *From,
+                                     const MachineBasicBlock *To,
+                                     NextUseDistance Value) const {
+    PathInfo &Slot = getOrInitPathInfo(From, To);
+    assert(!Slot.ShortestDistance.has_value());
+    Slot.ShortestDistance = Value;
+    return Value;
+  }
+
+  NextUseDistance
+  initializePathInfoShortestUnweightedDistance(const MachineBasicBlock *From,
+                                               const MachineBasicBlock *To,
+                                               NextUseDistance Value) const {
+    PathInfo &Slot = getOrInitPathInfo(From, To);
+    assert(!Slot.ShortestUnweightedDistance.has_value());
+    Slot.ShortestUnweightedDistance = Value;
+    return Value;
+  }
+
+  //----------------------------------------------------------------------------
+  // initialize*Paths
+  //----------------------------------------------------------------------------
+private:
+  void initializePaths(const SmallVector<Path> &ReachablePaths,
+                       const SmallVector<Path> &UnreachablePaths) const {
+    for (bool R : {true, false}) {
+      const auto &ToInit = R ? ReachablePaths : UnreachablePaths;
+      for (const Path &P : ToInit)
+        initializePathInfo(P, EdgeKind::None, R);
+    }
+  }
+
+  void
+  initializeForwardOnlyPaths(const SmallVector<Path> &ReachablePaths,
+                             const SmallVector<Path> &UnreachablePaths) const {
+    for (bool R : {true, false}) {
+      const auto &ToInit = R ? ReachablePaths : UnreachablePaths;
+      for (const Path &P : ToInit) {
+        PathInfo &Slot = getOrInitPathInfo(P.src(), P.dst());
+        assert(Slot.isForwardReachableUnset() || Slot.ForwardReachable == R);
+        Slot.ForwardReachable = R;
+      }
+    }
+  }
+
+  // Follow the control flow graph starting at the entry block until all blocks
+  // have been visited. Along the way, initialize the PathInfo for each edge
+  // traversed.
+  void initializeCfgPaths() {
+    Paths.clear();
+
+    enum VisitState { Undiscovered, Visiting, Finished };
+    DenseMap<const MachineBasicBlock *, VisitState> State;
+
+    SmallVector<const MachineBasicBlock *> Work{&MF->front()};
+    State[&MF->front()] = Undiscovered;
+
+    while (!Work.empty()) {
+      const MachineBasicBlock *Src = Work.back();
+      VisitState &SrcState = State[Src];
+
+      // A block may already be 'Finished' if it is reachable from multiple
+      // predecessors causing it to be pushed more than once while still
+      // 'Undiscovered'.
+      if (SrcState == Visiting || SrcState == Finished) {
+        Work.pop_back();
+        SrcState = Finished;
+        continue;
+      }
+
+      SrcState = Visiting;
+      for (const MachineBasicBlock *Dst : Src->successors()) {
+        const VisitState DstState = State.lookup(Dst);
+
+        EdgeKind EK;
+        if (DstState == Undiscovered) {
+          EK = EdgeKind::Forward;
+          Work.push_back(Dst);
+        } else if (DstState == Visiting) {
+          EK = EdgeKind::Back;
+        } else {
+          EK = EdgeKind::Forward;
+        }
+
+        Path P(Src, Dst);
+        assert(!Paths.contains(P));
+        initializePathInfo(P, EK, /*Reachable*/ true);
+      }
+    }
+
+    LLVM_DEBUG(dumpPaths());
+  }
+
+  //~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+  // Loop helpers
+  //~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+private:
+  static bool isStandAloneLoop(const MachineLoop *Loop) {
+    return Loop->getSubLoops().empty() && Loop->isOutermost();
+  }
+
+  static MachineLoop *findChildLoop(MachineLoop *const Parent,
+                                    MachineLoop *Descendant) {
+    for (MachineLoop *L = Descendant; L != Parent; L = L->getParentLoop()) {
+      if (L->getParentLoop() == Parent)
+        return L;
+    }
+    return nullptr;
+  }
+
+  // If loops 'A' and 'B' share a common parent loop, return that loop and the
+  // depth of 'A' relative to it. Otherwise return nullptr and the loop depth of
+  // 'A'.
+  static std::pair<MachineLoop *, unsigned>
+  findCommonParent(MachineLoop *A, const MachineLoop *const B) {
+    unsigned Depth = 0;
+    for (; A != nullptr; A = A->getParentLoop(), ++Depth) {
+      if (A->contains(B))
+        break;
+    }
+    return {A, Depth};
+  }
+
+  static const MachineBasicBlock *
+  getOutermostPreheader(const MachineLoop *Loop) {
+    return Loop ? Loop->getOutermostLoop()->getLoopPreheader() : nullptr;
+  }
+
+  static MachineBasicBlock *findChildPreheader(MachineLoop *const Parent,
+                                               MachineLoop *Descendant) {
+    MachineLoop *ChildLoop = findChildLoop(Parent, Descendant);
+    return ChildLoop ? ChildLoop->getLoopPreheader() : nullptr;
+  }
+
+  static const MachineBasicBlock *mbbForPhiOp(const MachineInstr *MI,
+                                              const MachineOperand *MO) {
+    return MI->isPHI() ? MI->getOperand(MO->getOperandNo() + 1).getMBB()
+                       : nullptr;
+  }
+
+  //~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+  // Calculate features
+  //~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+private:
+  InstrIdTy calcSize(const MachineBasicBlock *BB) const {
+    InstrIdTy Size = BB->size();
+    if (!Cfg.CountPhis)
+      Size -= std::distance(BB->begin(), BB->getFirstNonPHI());
+    return Size;
+  }
+
+  NextUseDistance calcWeightedSize(const MachineBasicBlock *From,
+                                   const MachineBasicBlock *To) const {
+    return NextUseDistance::fromSize(getSize(From),
+                                     getRelativeLoopDepth(From, To));
+  }
+
+  // Return the loop depth of 'From' relative to 'To'.
+  unsigned calcRelativeLoopDepth(const MachineBasicBlock *From,
+                                 const MachineBasicBlock *To) const {
+    MachineLoop *LoopFrom = MLI->getLoopFor(From);
+    MachineLoop *LoopTo = MLI->getLoopFor(To);
+
+    if (!LoopFrom)
+      return 0;
+
+    if (!LoopTo)
+      return LoopFrom->getLoopDepth();
+
+    if (LoopFrom->contains(LoopTo)) // covers LoopFrom == LoopTo
+      return 0;
+
+    if (LoopTo->contains(LoopFrom))
+      return LoopFrom->getLoopDepth() - LoopTo->getLoopDepth();
+
+    // Loops are siblings of some sort.
+    return findCommonParent(LoopFrom, LoopTo).second;
+  }
+
+  // Attempt to find a path from 'From' to 'To' using a depth first search. If
+  // 'ForwardOnly' is true, do not follow backedges. As a performance
+  // improvement, this may initialize reachable intermediate paths or paths we
+  // determine are unreachable.
+  bool calcIsReachable(const MachineBasicBlock *From,
+                       const MachineBasicBlock *To,
+                       bool ForwardOnly = false) const {
+    if (From == To && !MLI->getLoopFor(From))
+      return false;
+
+    if (!ForwardOnly && interBlockDistanceExists(From, To))
+      return true;
+
+    enum { VisitOp, PopOp };
+    using MBBOpPair = std::pair<const MachineBasicBlock *, int>;
+    SmallVector<MBBOpPair> Work{{From, VisitOp}};
+    DenseSet<const MachineBasicBlock *> Visited{From};
+
+    SmallVector<Path> IntermediatePath;
+    SmallVector<Path> Unreachable;
+
+    // Should be run at every function exit point.
+    auto Finally = [&](bool Reachable) {
+      // This is an optimization. For intermediate paths we found while
+      // calculating reachability for 'From' --> 'To', remember their
+      // reachability.
+      if (!Reachable) {
+        IntermediatePath.clear();
+        for (const MachineBasicBlock *MBB : Visited) {
+          if (MBB != From)
+            Unreachable.emplace_back(MBB, To);
+        }
+      }
+
+      if (ForwardOnly)
+        initializeForwardOnlyPaths(IntermediatePath, Unreachable);
+      else
+        initializePaths(IntermediatePath, Unreachable);
+
+      return Reachable;
+    };
+
+    while (!Work.empty()) {
+      auto [Current, Op] = Work.pop_back_val();
+
+      // Backtracking
+      if (Op == PopOp) {
+        IntermediatePath.pop_back();
+        if (ForwardOnly)
+          Unreachable.emplace_back(Current, To);
+        continue;
+      }
+
+      if (Current->succ_empty())
+        continue;
+
+      if (Current != From) {
+        IntermediatePath.emplace_back(Current, To);
+        Work.emplace_back(Current, PopOp);
+      }
+
+      for (const MachineBasicBlock *Succ : Current->successors()) {
+        if (ForwardOnly && isBackedge(Current, Succ))
+          continue;
+
+        if (Succ == To)
+          return Finally(true);
+
+        if (auto CachedReachable = isMaybeReachable(Succ, To, ForwardOnly)) {
+          if (CachedReachable.value())
+            return Finally(true);
+          Visited.insert(Succ);
+          continue;
+        }
+
+        if (Visited.insert(Succ).second)
+          Work.emplace_back(Succ, VisitOp);
+      }
+    }
+
+    return Finally(false);
+  }
+
+  //----------------------------------------------------------------------------
+  // Inter-block distance - the weighted and unweighted cost (i.e. "distance")
+  // to travel from one MachineBasicBlock to another.
+  //
+  // Values are pre-computed and stored in 'InterBlockDistances' using a
+  // backwards data-flow algorithm similar to the one described in 4.1 of a
+  // "Register Spilling and Live-Range Splitting for SSA-Form Programs" by
+  // Matthias Braun and Sebastian Hack, CC'09. This replaced a prior
+  // implementation based on Dijkstra's shortest path algorithm.
+  //----------------------------------------------------------------------------
+private:
+  struct InterBlockDistance {
+    NextUseDistance Weighted;
+    NextUseDistance Unweighted;
+    InterBlockDistance() : Weighted(-1), Unweighted(-1) {}
+    InterBlockDistance(NextUseDistance W, NextUseDistance UW)
+        : Weighted(W), Unweighted(UW) {}
+    bool operator==(const InterBlockDistance &Other) const {
+      return Weighted == Other.Weighted && Unweighted == Other.Unweighted;
+    }
+    bool operator!=(const InterBlockDistance &Other) const {
+      return !(*this == Other);
+    }
+
+    void print(raw_ostream &OS) const {
+      OS << "{W=";
+      Weighted.print(OS);
+      OS << " U=";
+      Unweighted.print(OS);
+      OS << "}";
+    }
+
+    LLVM_DUMP_METHOD void dump() const {
+      print(dbgs());
+      dbgs() << '\n';
+    }
+  };
+  using InterBlockDistanceMap =
+      DenseMap<unsigned, DenseMap<unsigned, InterBlockDistance>>;
+  InterBlockDistanceMap InterBlockDistances;
+
+  void initializeInterBlockDistances() {
+    InterBlockDistanceMap Distances;
+
+    bool Changed;
+    do {
+      Changed = false;
+      for (const MachineBasicBlock *MBB : post_order(MF)) {
+        unsigned MBBNum = MBB->getNumber();
+
+        // Save previous state for convergence check
+        InterBlockDistanceMap::mapped_type Prev = std::move(Distances[MBBNum]);
+        InterBlockDistanceMap::mapped_type Curr;
+        Curr.reserve(Prev.size());
+
+        // Direct successors are distance 0 by definition: no instructions are
+        // executed between exiting MBB and entering Succ.
+        for (const MachineBasicBlock *Succ : MBB->successors())
+          Curr[Succ->getNumber()] = InterBlockDistance(0, 0);
+
+        // Propagate further destinations through each successor.
+        for (const MachineBasicBlock *Succ : MBB->successors()) {
+          unsigned SuccNum = Succ->getNumber();
+          const unsigned UnweightedSize{getSize(Succ)};
+
+          for (const auto &[DestBlockNum, DestDist] : Distances[SuccNum]) {
+            // MBB -> MBB is considered unreachable (getInterBlockDistance
+            // asserts From != To).
+            if (DestBlockNum == MBBNum)
+              continue;
+
+            const MachineBasicBlock *DestMBB =
+                MF->getBlockNumbered(DestBlockNum);
+
+            const NextUseDistance UnweightedDist{UnweightedSize +
+                                                 DestDist.Unweighted};
+
+            unsigned SuccToDestLoopDepth = calcRelativeLoopDepth(Succ, DestMBB);
+
+            const NextUseDistance WeightedDist =
+                DestDist.Weighted +
+                NextUseDistance::fromSize(UnweightedSize, SuccToDestLoopDepth);
+
+            // Insert or update distances (take minimum)
+            auto [I, First] =
+                Curr.try_emplace(DestBlockNum, WeightedDist, UnweightedDist);
+            if (!First) {
+              InterBlockDistance &Slot = I->second;
+              Slot.Weighted = min(Slot.Weighted, WeightedDist);
+              Slot.Unweighted = min(Slot.Unweighted, UnweightedDist);
+            }
+          }
+        }
+        Changed |= (Prev != Curr);
+        Distances[MBBNum] = std::move(Curr);
+      }
+    } while (Changed);
+
+    InterBlockDistances = std::move(Distances);
+    LLVM_DEBUG(dumpInterBlockDistances());
+  }
+
+  const InterBlockDistance *
+  getInterBlockDistanceMapValue(const MachineBasicBlock *From,
+                                const MachineBasicBlock *To) const {
+    auto I = InterBlockDistances.find(From->getNumber());
+    if (I == InterBlockDistances.end())
+      return nullptr;
+    const InterBlockDistanceMap::mapped_type &FromSlot = I->second;
+    auto J = FromSlot.find(To->getNumber());
+    return J == FromSlot.end() ? nullptr : &J->second;
+  }
+
+  bool interBlockDistanceExists(const MachineBasicBlock *From,
+                                const MachineBasicBlock *To) const {
+    return getInterBlockDistanceMapValue(From, To);
+  }
+
+  NextUseDistance getInterBlockDistance(const MachineBasicBlock *From,
+                                        const MachineBasicBlock *To,
+                                        bool Unweighted) const {
+
+    assert(From != To && "The basic blocks should be different.");
+    if (!From || !To)
+      return NextUseDistance::unreachable();
+
+    if (Cfg.ForwardOnly && !isForwardReachable(From, To))
+      return NextUseDistance::unreachable();
+
+    const InterBlockDistance *BD = getInterBlockDistanceMapValue(From, To);
+    if (!BD)
+      return NextUseDistance::unreachable();
+
+    return Unweighted ? BD->Unweighted : BD->Weighted;
+  }
+
+  NextUseDistance
+  getWeightedInterBlockDistance(const MachineBasicBlock *From,
+                                const MachineBasicBlock *To) const {
+    return getInterBlockDistance(From, To, false);
+  }
+
+  NextUseDistance
+  getUnweightedInterBlockDistance(const MachineBasicBlock *From,
+                                  const MachineBasicBlock *To) const {
+    return getInterBlockDistance(From, To, true);
+  }
+
+  //~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+  // Feature getters. Use cached results if available. If not calculate.
+  //~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+private:
+  InstrIdTy getSize(const MachineBasicBlock *BB) const {
+    return pathInfoFor(BB, BB).Size;
+  }
+
+  bool isReachable(const MachineBasicBlock *From,
+                   const MachineBasicBlock *To) const {
+    return pathInfoFor(From, To).Reachable;
+  }
+
+  bool isReachableOrSame(const MachineBasicBlock *From,
+                         const MachineBasicBlock *To) const {
+    return From == To || pathInfoFor(From, To).Reachable;
+  }
+
+  bool isForwardReachable(const MachineBasicBlock *From,
+                          const MachineBasicBlock *To) const {
+    const PathInfo &PI = pathInfoFor(From, To);
+    if (PI.isForwardReachableSet())
+      return PI.isForwardReachable();
+
+    return initializePathInfoForwardReachable(
+        From, To,
+        PI.Reachable && calcIsReachable(From, To, /*ForwardOnly*/ true));
+  }
+
+  // Return true/false if we know that 'To' is reachable or not from
+  // 'From'. Otherwise return 'std::nullopt'.
+  std::optional<bool> isMaybeReachable(const MachineBasicBlock *From,
+                                       const MachineBasicBlock *To,
+                                       bool ForwardOnly) const {
+    const PathInfo *PI = maybePathInfoFor(From, To);
+    if (!PI)
+      return std::nullopt;
+
+    if (ForwardOnly) {
+      if (PI->isForwardReachable())
+        return true;
+
+      if (PI->isNotForwardReachable())
+        return false;
+      return std::nullopt;
+    }
+    return PI->Reachable;
+  }
+
+  bool isBackedge(const MachineBasicBlock *From,
+                  const MachineBasicBlock *To) const {
+    return pathInfoFor(From, To).isBackedge();
+  }
+
+  // Can be used as a substitute for DT->dominates(A, B) if A and B are in the
+  // same basic block.
+  bool instrsAreInOrder(const MachineInstr *A, const MachineInstr *B) const {
----------------
arsenm wrote:

You can do this by comparing SlotIndexes, which you have already? 

https://github.com/llvm/llvm-project/pull/178873


More information about the llvm-commits mailing list