[llvm] [llvm][CodeGen][AArch64] Allow the WindowScheduler to pipeline SUBS+Bcc-terminated loops (PR #191587)

via llvm-commits llvm-commits at lists.llvm.org
Fri Apr 10 19:10:47 PDT 2026


llvmbot wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-aarch64

Author: Jon Roelofs (jroelofs)

<details>
<summary>Changes</summary>

Previously they were ignored because SUBS has an implicit phys reg def, but we can relax that a bit when NZCV is known to only be used by the loop terminator as the dependence will naturally constrain the SUBS to stage 0 along with the branch.

---
Full diff: https://github.com/llvm/llvm-project/pull/191587.diff


7 Files Affected:

- (modified) llvm/include/llvm/CodeGen/TargetInstrInfo.h (+14-2) 
- (modified) llvm/lib/CodeGen/WindowScheduler.cpp (+2-7) 
- (modified) llvm/lib/Target/AArch64/AArch64InstrInfo.cpp (+21-4) 
- (modified) llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp (+3-2) 
- (modified) llvm/lib/Target/PowerPC/PPCInstrInfo.cpp (+3-2) 
- (modified) llvm/lib/Target/RISCV/RISCVInstrInfo.cpp (+1-1) 
- (added) llvm/test/CodeGen/AArch64/sms-subs-nzcv-private.mir (+126) 


``````````diff
diff --git a/llvm/include/llvm/CodeGen/TargetInstrInfo.h b/llvm/include/llvm/CodeGen/TargetInstrInfo.h
index cd5561e57d033..c2d4fc7208060 100644
--- a/llvm/include/llvm/CodeGen/TargetInstrInfo.h
+++ b/llvm/include/llvm/CodeGen/TargetInstrInfo.h
@@ -822,8 +822,20 @@ class LLVM_ABI TargetInstrInfo : public MCInstrInfo {
     virtual ~PipelinerLoopInfo();
     /// Return true if the given instruction should not be pipelined and should
     /// be ignored. An example could be a loop comparison, or induction variable
-    /// update with no users being pipelined.
-    virtual bool shouldIgnoreForPipelining(const MachineInstr *MI) const = 0;
+    /// update with no users being pipelined. By default we ignore instructions
+    /// with physical register defs because the pipeliner cannot reason about
+    /// physical register lifetimes.
+    virtual bool shouldIgnoreForPipelining(const MachineInstr *MI) const {
+      for (const auto &MO : MI->all_defs())
+        if (MO.isReg() && MO.getReg().isPhysical()) {
+          DEBUG_WITH_TYPE(
+              "pipeliner",
+              dbgs() << "MI defines a physical reg; ignoring for pipelining:\n"
+                     << *MI << "\n");
+          return true;
+        }
+      return false;
+    }
 
     /// Return true if the proposed schedule should used.  Otherwise return
     /// false to not pipeline the loop. This function should be used to ensure
diff --git a/llvm/lib/CodeGen/WindowScheduler.cpp b/llvm/lib/CodeGen/WindowScheduler.cpp
index 2492dfc3ca553..aaac27dff65b8 100644
--- a/llvm/lib/CodeGen/WindowScheduler.cpp
+++ b/llvm/lib/CodeGen/WindowScheduler.cpp
@@ -229,15 +229,10 @@ bool WindowScheduler::initialize() {
     }
     if (PLI->shouldIgnoreForPipelining(&MI)) {
       LLVM_DEBUG(dbgs() << "Special MI defined by target is not allowed in "
-                           "window scheduling!\n");
+                           "window scheduling:\n"
+                        << MI << "\n");
       return false;
     }
-    for (auto &Def : MI.all_defs())
-      if (Def.isReg() && Def.getReg().isPhysical()) {
-        LLVM_DEBUG(dbgs() << "Physical registers are not supported in "
-                             "window scheduling!\n");
-        return false;
-      }
   }
   if (SchedInstrNum <= WindowRegionLimit) {
     LLVM_DEBUG(dbgs() << "There are too few MIs in the window region!\n");
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
index 4094526574d7a..3235a2f3881ad 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
@@ -11410,6 +11410,10 @@ class AArch64PipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
   /// The normalized condition used by createTripCountGreaterCondition()
   SmallVector<MachineOperand, 4> Cond;
 
+  /// True iff \p CondBranch is the only use of \p Comp's NZCV def, making it
+  /// private to the back-edge of the loop.
+  bool NZCVIsBackedgePrivate;
+
 public:
   AArch64PipelinerLoopInfo(MachineBasicBlock *LoopBB, MachineInstr *CondBranch,
                            MachineInstr *Comp, unsigned CompCounterOprNum,
@@ -11422,12 +11426,25 @@ class AArch64PipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
         LoopBB(LoopBB), CondBranch(CondBranch), Comp(Comp),
         CompCounterOprNum(CompCounterOprNum), Update(Update),
         UpdateCounterOprNum(UpdateCounterOprNum), Init(Init),
-        IsUpdatePriorComp(IsUpdatePriorComp), Cond(Cond.begin(), Cond.end()) {}
+        IsUpdatePriorComp(IsUpdatePriorComp), Cond(Cond.begin(), Cond.end()),
+        NZCVIsBackedgePrivate(true) {
+    for (const MachineOperand &MO : MRI.use_nodbg_operands(AArch64::NZCV)) {
+      if (MO.getParent()->getParent() == LoopBB &&
+          MO.getParent() != CondBranch) {
+        NZCVIsBackedgePrivate = false;
+        break;
+      }
+    }
+  }
 
   bool shouldIgnoreForPipelining(const MachineInstr *MI) const override {
-    // Make the instructions for loop control be placed in stage 0.
-    // The predecessors of Comp are considered by the caller.
-    return MI == Comp;
+    if (MI == Comp) {
+      // If Comp's NZCV is only used by CondBranch, the loop-carried counter
+      // dependency naturally constrains Comp to stage 0, and no explicit
+      // pinning is needed.
+      return !NZCVIsBackedgePrivate;
+    }
+    return PipelinerLoopInfo::shouldIgnoreForPipelining(MI);
   }
 
   std::optional<bool> createTripCountGreaterCondition(
diff --git a/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp b/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
index 732256f61556f..cd24c32221f92 100644
--- a/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
+++ b/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
@@ -748,8 +748,9 @@ class HexagonPipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
   }
 
   bool shouldIgnoreForPipelining(const MachineInstr *MI) const override {
-    // Only ignore the terminator.
-    return MI == EndLoop;
+    if (MI == EndLoop)
+      return true;
+    return PipelinerLoopInfo::shouldIgnoreForPipelining(MI);
   }
 
   std::optional<bool> createTripCountGreaterCondition(
diff --git a/llvm/lib/Target/PowerPC/PPCInstrInfo.cpp b/llvm/lib/Target/PowerPC/PPCInstrInfo.cpp
index 6d95314d4b019..fec09ca6d8017 100644
--- a/llvm/lib/Target/PowerPC/PPCInstrInfo.cpp
+++ b/llvm/lib/Target/PowerPC/PPCInstrInfo.cpp
@@ -5753,8 +5753,9 @@ class PPCPipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
   }
 
   bool shouldIgnoreForPipelining(const MachineInstr *MI) const override {
-    // Only ignore the terminator.
-    return MI == EndLoop;
+    if (MI == EndLoop)
+      return true;
+    return PipelinerLoopInfo::shouldIgnoreForPipelining(MI);
   }
 
   std::optional<bool> createTripCountGreaterCondition(
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfo.cpp b/llvm/lib/Target/RISCV/RISCVInstrInfo.cpp
index db559f4949904..880b2ea909fb5 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfo.cpp
@@ -5251,7 +5251,7 @@ class RISCVPipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
       return true;
     if (RHS && MI == RHS)
       return true;
-    return false;
+    return PipelinerLoopInfo::shouldIgnoreForPipelining(MI);
   }
 
   std::optional<bool> createTripCountGreaterCondition(
diff --git a/llvm/test/CodeGen/AArch64/sms-subs-nzcv-private.mir b/llvm/test/CodeGen/AArch64/sms-subs-nzcv-private.mir
new file mode 100644
index 0000000000000..3c5dc1b23b2f7
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sms-subs-nzcv-private.mir
@@ -0,0 +1,126 @@
+# RUN: llc --verify-machineinstrs -mtriple=aarch64-apple-macosx -mcpu=apple-m1 \
+# RUN:   -run-pass pipeliner -debug-only=pipeliner -o /dev/null %s 2>&1 | \
+# RUN:   FileCheck %s
+# RUN: llc --verify-machineinstrs -mtriple=aarch64-apple-macosx -mcpu=apple-m1 \
+# RUN:   -run-pass pipeliner -debug-only=pipeliner -window-sched=force -o /dev/null %s 2>&1 | \
+# RUN:   FileCheck %s
+
+# REQUIRES: asserts
+
+# Check that we allow pipelining of loops with a phys reg def of NZCV, but only
+# when one use is private to the condition on the loop backedge. This allows
+# pipelining on a common class of loops on aarch64 where the exit condition is
+# predicated on the decrement of the trip count.
+
+# CHECK:      Special MI defined by target is not allowed in window scheduling:
+# CHECK-NEXT: %{{.*}}:gpr64 = nsw SUBSXri %{{.*}}:gpr64sp, 1, 0, implicit-def $nzcv
+# CHECK-NOT:  Special MI defined by target is not allowed in window scheduling:
+
+--- |
+  define void @private_nzcv(i64 %n, float %fa, ptr noalias %x, ptr noalias %y) {
+  entry:
+    br i1 false, label %preheader, label %exit
+  preheader:
+    br label %for.body
+  for.body:
+    br i1 false, label %exit, label %for.body
+  exit:
+    ret void
+  }
+  define void @shared_nzcv(i64 %n, float %fa, ptr %fx) {
+  entry:
+    br i1 false, label %preheader, label %exit
+  preheader:
+    br label %for.body
+  for.body:
+    br i1 false, label %exit, label %for.body
+  exit:
+    ret void
+  }
+...
+---
+name:            private_nzcv
+tracksRegLiveness: true
+liveins:
+  - { reg: '$x0', virtual-reg: '%3' }
+  - { reg: '$s0', virtual-reg: '%4' }
+  - { reg: '$x1', virtual-reg: '%5' }
+  - { reg: '$x2', virtual-reg: '%6' }
+body:             |
+  bb.0.entry:
+    successors: %bb.1(0x50000000), %bb.2(0x30000000)
+    liveins: $x0, $s0, $x1, $x2
+
+    %3:gpr64sp = COPY $x0
+    %4:fpr32 = COPY $s0
+    %5:gpr64sp = COPY $x1
+    %6:gpr64sp = COPY $x2
+    dead $xzr = SUBSXri %3, 1, 0, implicit-def $nzcv
+    Bcc 3, %bb.2, implicit $nzcv
+    B %bb.1
+
+  bb.1.preheader:
+    %0:gpr64sp = COPY %3
+    B %bb.3
+
+  bb.2.exit:
+    RET_ReallyLR
+
+  bb.3.for.body:
+    successors: %bb.2(0x04000000), %bb.3(0x7c000000)
+
+    %1:gpr64sp = PHI %0, %bb.1, %7, %bb.3
+    %8:gpr64sp = PHI %5, %bb.1, %9, %bb.3
+    %10:gpr64sp = PHI %6, %bb.1, %11, %bb.3
+    early-clobber %9:gpr64sp, %12:fpr32 = LDRSpost %8, 4 :: (load (s32) from %ir.x)
+    %13:fpr32 = nofpexcept FMULSrr %4, killed %12, implicit $fpcr
+    %14:fpr32 = nofpexcept FMULSrr %4, killed %13, implicit $fpcr
+    early-clobber %11:gpr64sp = STRSpost killed %14, %10, 4 :: (store (s32) into %ir.y)
+    %15:gpr64 = nsw SUBSXri %1, 1, 0, implicit-def $nzcv
+    %7:gpr64all = COPY %15
+    Bcc 1, %bb.3, implicit $nzcv
+    B %bb.2
+
+...
+---
+name:            shared_nzcv
+tracksRegLiveness: true
+liveins:
+  - { reg: '$x0', virtual-reg: '%3' }
+  - { reg: '$s0', virtual-reg: '%4' }
+  - { reg: '$x1', virtual-reg: '%5' }
+body:             |
+  bb.0.entry:
+    successors: %bb.1(0x50000000), %bb.2(0x30000000)
+    liveins: $x0, $s0, $x1
+
+    %3:gpr64sp = COPY $x0
+    %4:fpr32 = COPY $s0
+    %5:gpr64sp = COPY $x1
+    dead $xzr = SUBSXri %3, 1, 0, implicit-def $nzcv
+    Bcc 3, %bb.2, implicit $nzcv
+    B %bb.1
+
+  bb.1.preheader:
+    %0:gpr64sp = COPY %3
+    B %bb.3
+
+  bb.2.exit:
+    RET_ReallyLR
+
+  bb.3.for.body:
+    successors: %bb.2(0x04000000), %bb.3(0x7c000000)
+
+    %1:gpr64sp = PHI %0, %bb.1, %6, %bb.3
+    %9:gpr64sp = PHI %5, %bb.1, %10, %bb.3
+    %11:fpr32 = LDRSui %9, 0 :: (load (s32))
+    %15:gpr64 = nsw SUBSXri %1, 1, 0, implicit-def $nzcv
+    %16:fpr32 = FCSELSrrr %4, killed %11, 1, implicit $nzcv
+    %12:fpr32 = nofpexcept FMULSrr %4, killed %16, implicit $fpcr
+    %14:fpr32 = nofpexcept FMULSrr %4, killed %12, implicit $fpcr
+    early-clobber %10:gpr64sp = STRSpost killed %14, killed %9, 4 :: (store (s32))
+    %6:gpr64all = COPY %15
+    Bcc 1, %bb.3, implicit $nzcv
+    B %bb.2
+
+...

``````````

</details>


https://github.com/llvm/llvm-project/pull/191587


More information about the llvm-commits mailing list