[llvm] [llvm][CodeGen][AArch64] Allow the WindowScheduler to pipeline SUBS+Bcc-terminated loops (PR #191587)

Jon Roelofs via llvm-commits llvm-commits at lists.llvm.org
Sat Apr 11 14:18:12 PDT 2026


https://github.com/jroelofs updated https://github.com/llvm/llvm-project/pull/191587

>From 29034f21a85d4bee5e2cbf585a5aca29e9657f9a Mon Sep 17 00:00:00 2001
From: Jon Roelofs <jonathan_roelofs at apple.com>
Date: Fri, 10 Apr 2026 16:48:32 -0700
Subject: [PATCH] [llvm][CodeGen][AArch64] Allow the WindowScheduler to
 pipeline SUBS+Bcc-terminated loops

Previously they were ignored because SUBS has an implicit phys reg def, but we
can relax that a bit when NZCV is known to only be used by the loop terminator
as the dependence will naturally constrain the SUBS to stage 0 along with the
branch.
---
 llvm/include/llvm/CodeGen/TargetInstrInfo.h   |   7 +
 llvm/lib/CodeGen/WindowScheduler.cpp          |   6 +-
 llvm/lib/Target/AArch64/AArch64InstrInfo.cpp  |  46 +++++--
 .../aarch64-swp-ws-live-intervals-1.mir       |  51 ++++---
 .../CodeGen/AArch64/sms-subs-nzcv-private.mir | 126 ++++++++++++++++++
 5 files changed, 199 insertions(+), 37 deletions(-)
 create mode 100644 llvm/test/CodeGen/AArch64/sms-subs-nzcv-private.mir

diff --git a/llvm/include/llvm/CodeGen/TargetInstrInfo.h b/llvm/include/llvm/CodeGen/TargetInstrInfo.h
index cd5561e57d033..5239ab924026a 100644
--- a/llvm/include/llvm/CodeGen/TargetInstrInfo.h
+++ b/llvm/include/llvm/CodeGen/TargetInstrInfo.h
@@ -825,6 +825,13 @@ class LLVM_ABI TargetInstrInfo : public MCInstrInfo {
     /// update with no users being pipelined.
     virtual bool shouldIgnoreForPipelining(const MachineInstr *MI) const = 0;
 
+    /// Return true if the given instruction's physical register def is safe for
+    /// window scheduling. By default, physical register defs are not allowed.
+    virtual bool
+    allowPhysRegDefInWindowScheduler(const MachineInstr *MI) const {
+      return false;
+    }
+
     /// Return true if the proposed schedule should used.  Otherwise return
     /// false to not pipeline the loop. This function should be used to ensure
     /// that pipelined loops meet target-specific quality heuristics.
diff --git a/llvm/lib/CodeGen/WindowScheduler.cpp b/llvm/lib/CodeGen/WindowScheduler.cpp
index 2492dfc3ca553..b76a4a957398c 100644
--- a/llvm/lib/CodeGen/WindowScheduler.cpp
+++ b/llvm/lib/CodeGen/WindowScheduler.cpp
@@ -229,11 +229,13 @@ bool WindowScheduler::initialize() {
     }
     if (PLI->shouldIgnoreForPipelining(&MI)) {
       LLVM_DEBUG(dbgs() << "Special MI defined by target is not allowed in "
-                           "window scheduling!\n");
+                           "window scheduling:\n"
+                        << MI);
       return false;
     }
     for (auto &Def : MI.all_defs())
-      if (Def.isReg() && Def.getReg().isPhysical()) {
+      if (Def.isReg() && Def.getReg().isPhysical() &&
+          !PLI->allowPhysRegDefInWindowScheduler(&MI)) {
         LLVM_DEBUG(dbgs() << "Physical registers are not supported in "
                              "window scheduling!\n");
         return false;
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
index 4094526574d7a..5f183c4585340 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
@@ -11386,7 +11386,7 @@ AArch64InstrInfo::probedStackAlloc(MachineBasicBlock::iterator MBBI,
 namespace {
 class AArch64PipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
   MachineFunction *MF;
-  const TargetInstrInfo *TII;
+  const AArch64InstrInfo *TII;
   const TargetRegisterInfo *TRI;
   MachineRegisterInfo &MRI;
 
@@ -11410,6 +11410,10 @@ class AArch64PipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
   /// The normalized condition used by createTripCountGreaterCondition()
   SmallVector<MachineOperand, 4> Cond;
 
+  /// True iff \p CondBranch is the only use of \p Comp's NZCV def, making it
+  /// private to the back-edge of the loop.
+  bool IsNZCVBackedgePrivate;
+
 public:
   AArch64PipelinerLoopInfo(MachineBasicBlock *LoopBB, MachineInstr *CondBranch,
                            MachineInstr *Comp, unsigned CompCounterOprNum,
@@ -11417,17 +11421,41 @@ class AArch64PipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
                            Register Init, bool IsUpdatePriorComp,
                            const SmallVectorImpl<MachineOperand> &Cond)
       : MF(Comp->getParent()->getParent()),
-        TII(MF->getSubtarget().getInstrInfo()),
-        TRI(MF->getSubtarget().getRegisterInfo()), MRI(MF->getRegInfo()),
-        LoopBB(LoopBB), CondBranch(CondBranch), Comp(Comp),
-        CompCounterOprNum(CompCounterOprNum), Update(Update),
+        TII(MF->getSubtarget<AArch64Subtarget>().getInstrInfo()),
+        TRI(MF->getSubtarget<AArch64Subtarget>().getRegisterInfo()),
+        MRI(MF->getRegInfo()), LoopBB(LoopBB), CondBranch(CondBranch),
+        Comp(Comp), CompCounterOprNum(CompCounterOprNum), Update(Update),
         UpdateCounterOprNum(UpdateCounterOprNum), Init(Init),
-        IsUpdatePriorComp(IsUpdatePriorComp), Cond(Cond.begin(), Cond.end()) {}
+        IsUpdatePriorComp(IsUpdatePriorComp), Cond(Cond.begin(), Cond.end()),
+        IsNZCVBackedgePrivate(isNZCVBackedgePrivate()) {}
+
+  bool isNZCVBackedgePrivate() {
+    // NZCV is private to the backedge if CondBranch is the only instruction in
+    // LoopBB that uses it.
+    for (const MachineInstr &MI : *LoopBB) {
+      if (&MI == CondBranch)
+        continue;
+      for (const MachineOperand &MO : MI.operands()) {
+        if (MO.isReg() && MO.isUse() && MO.getReg() == AArch64::NZCV) {
+          return false;
+        }
+      }
+    }
+    return true;
+  }
 
   bool shouldIgnoreForPipelining(const MachineInstr *MI) const override {
-    // Make the instructions for loop control be placed in stage 0.
-    // The predecessors of Comp are considered by the caller.
-    return MI == Comp;
+    if (MI != Comp)
+      return false;
+    return !IsNZCVBackedgePrivate || TII->isWhileOpcode(Comp->getOpcode());
+  }
+
+  bool allowPhysRegDefInWindowScheduler(const MachineInstr *MI) const override {
+    // SUBS/ADDS with NZCV private to the backedge: the loop-carried counter
+    // dependency keeps Comp in stage 0, so $nzcv never crosses a stage
+    // boundary.
+    return MI == Comp && IsNZCVBackedgePrivate &&
+           !TII->isWhileOpcode(Comp->getOpcode());
   }
 
   std::optional<bool> createTripCountGreaterCondition(
diff --git a/llvm/test/CodeGen/AArch64/aarch64-swp-ws-live-intervals-1.mir b/llvm/test/CodeGen/AArch64/aarch64-swp-ws-live-intervals-1.mir
index 851944b1beec0..f4aa86093bfc8 100644
--- a/llvm/test/CodeGen/AArch64/aarch64-swp-ws-live-intervals-1.mir
+++ b/llvm/test/CodeGen/AArch64/aarch64-swp-ws-live-intervals-1.mir
@@ -22,7 +22,6 @@ body:             |
   ; CHECK-NEXT: {{  $}}
   ; CHECK-NEXT:   [[LDRDui:%[0-9]+]]:fpr64 = LDRDui [[COPY1]], 0
   ; CHECK-NEXT:   [[SUBSXri:%[0-9]+]]:gpr64 = nsw SUBSXri [[SUBREG_TO_REG]], 1, 0, implicit-def $nzcv
-  ; CHECK-NEXT:   [[COPY3:%[0-9]+]]:gpr64all = COPY [[SUBSXri]]
   ; CHECK-NEXT:   [[SUBREG_TO_REG1:%[0-9]+]]:gpr64all = SUBREG_TO_REG [[MOVi32imm]], %subreg.sub_32
   ; CHECK-NEXT:   Bcc 0, %bb.15, implicit $nzcv
   ; CHECK-NEXT:   B %bb.14
@@ -30,21 +29,21 @@ body:             |
   ; CHECK-NEXT: bb.14:
   ; CHECK-NEXT:   successors: %bb.15(0x04000000), %bb.14(0x7c000000)
   ; CHECK-NEXT: {{  $}}
-  ; CHECK-NEXT:   [[PHI:%[0-9]+]]:gpr64sp = PHI [[COPY3]], %bb.13, %73, %bb.14
-  ; CHECK-NEXT:   [[PHI1:%[0-9]+]]:gpr64 = PHI [[SUBREG_TO_REG1]], %bb.13, %72, %bb.14
-  ; CHECK-NEXT:   [[PHI2:%[0-9]+]]:fpr64 = PHI [[LDRDui]], %bb.13, %70, %bb.14
-  ; CHECK-NEXT:   STRDui [[PHI2]], [[COPY]], 0
+  ; CHECK-NEXT:   [[PHI:%[0-9]+]]:gpr64 = PHI [[SUBREG_TO_REG1]], %bb.13, %72, %bb.14
+  ; CHECK-NEXT:   [[PHI1:%[0-9]+]]:fpr64 = PHI [[LDRDui]], %bb.13, %70, %bb.14
+  ; CHECK-NEXT:   [[PHI2:%[0-9]+]]:gpr64 = PHI [[SUBSXri]], %bb.13, %71, %bb.14
+  ; CHECK-NEXT:   [[COPY3:%[0-9]+]]:gpr64sp = COPY [[PHI2]]
+  ; CHECK-NEXT:   STRDui [[PHI1]], [[COPY]], 0
   ; CHECK-NEXT:   [[LDRDui1:%[0-9]+]]:fpr64 = LDRDui [[COPY1]], 0
-  ; CHECK-NEXT:   [[SUBSXri1:%[0-9]+]]:gpr64 = nsw SUBSXri [[PHI]], 1, 0, implicit-def $nzcv
+  ; CHECK-NEXT:   [[SUBSXri1:%[0-9]+]]:gpr64 = nsw SUBSXri [[COPY3]], 1, 0, implicit-def $nzcv
   ; CHECK-NEXT:   [[SUBREG_TO_REG2:%[0-9]+]]:gpr64all = SUBREG_TO_REG [[MOVi32imm]], %subreg.sub_32
-  ; CHECK-NEXT:   [[COPY4:%[0-9]+]]:gpr64all = COPY [[SUBSXri1]]
   ; CHECK-NEXT:   Bcc 1, %bb.14, implicit $nzcv
   ; CHECK-NEXT:   B %bb.15
   ; CHECK-NEXT: {{  $}}
   ; CHECK-NEXT: bb.15:
   ; CHECK-NEXT:   successors: %bb.2(0x80000000)
   ; CHECK-NEXT: {{  $}}
-  ; CHECK-NEXT:   [[PHI3:%[0-9]+]]:gpr64 = PHI [[COPY2]], %bb.13, [[PHI1]], %bb.14
+  ; CHECK-NEXT:   [[PHI3:%[0-9]+]]:gpr64 = PHI [[COPY2]], %bb.13, [[PHI]], %bb.14
   ; CHECK-NEXT:   [[PHI4:%[0-9]+]]:fpr64 = PHI [[LDRDui]], %bb.13, [[LDRDui1]], %bb.14
   ; CHECK-NEXT:   STRDui [[PHI4]], [[COPY]], 0
   ; CHECK-NEXT:   B %bb.2
@@ -67,14 +66,14 @@ body:             |
   ; CHECK-NEXT:   successors: %bb.7(0x80000000)
   ; CHECK-NEXT: {{  $}}
   ; CHECK-NEXT:   [[PHI5:%[0-9]+]]:fpr64 = PHI [[FMOVD0_]], %bb.4, %62, %bb.12
-  ; CHECK-NEXT:   [[COPY5:%[0-9]+]]:gpr64all = COPY $xzr
-  ; CHECK-NEXT:   [[COPY6:%[0-9]+]]:gpr64common = COPY [[COPY5]]
+  ; CHECK-NEXT:   [[COPY4:%[0-9]+]]:gpr64all = COPY $xzr
+  ; CHECK-NEXT:   [[COPY5:%[0-9]+]]:gpr64common = COPY [[COPY4]]
   ; CHECK-NEXT:   B %bb.7
   ; CHECK-NEXT: {{  $}}
   ; CHECK-NEXT: bb.7:
   ; CHECK-NEXT:   successors: %bb.8(0x40000000), %bb.11(0x40000000)
   ; CHECK-NEXT: {{  $}}
-  ; CHECK-NEXT:   [[ADDSXri:%[0-9]+]]:gpr64common = ADDSXri [[COPY6]], 1, 0, implicit-def $nzcv
+  ; CHECK-NEXT:   [[ADDSXri:%[0-9]+]]:gpr64common = ADDSXri [[COPY5]], 1, 0, implicit-def $nzcv
   ; CHECK-NEXT:   [[CSINCXr:%[0-9]+]]:gpr64common = CSINCXr $xzr, $xzr, 3, implicit $nzcv
   ; CHECK-NEXT:   [[ADDSXri1:%[0-9]+]]:gpr64common = ADDSXri [[ADDSXri]], 1, 0, implicit-def $nzcv
   ; CHECK-NEXT:   [[CSINCXr1:%[0-9]+]]:gpr64common = CSINCXr [[CSINCXr]], [[CSINCXr]], 3, implicit $nzcv
@@ -89,35 +88,35 @@ body:             |
   ; CHECK-NEXT: bb.8:
   ; CHECK-NEXT:   successors: %bb.9(0x80000000)
   ; CHECK-NEXT: {{  $}}
-  ; CHECK-NEXT:   [[ADDSXri4:%[0-9]+]]:gpr64 = ADDSXri [[COPY6]], 1, 0, implicit-def $nzcv
-  ; CHECK-NEXT:   [[UBFMXri:%[0-9]+]]:gpr64common = UBFMXri [[COPY6]], 61, 60
-  ; CHECK-NEXT:   [[COPY7:%[0-9]+]]:gpr64common = COPY [[ADDSXri4]]
+  ; CHECK-NEXT:   [[ADDSXri4:%[0-9]+]]:gpr64 = ADDSXri [[COPY5]], 1, 0, implicit-def $nzcv
+  ; CHECK-NEXT:   [[UBFMXri:%[0-9]+]]:gpr64common = UBFMXri [[COPY5]], 61, 60
+  ; CHECK-NEXT:   [[COPY6:%[0-9]+]]:gpr64common = COPY [[ADDSXri4]]
   ; CHECK-NEXT:   [[LDRDui2:%[0-9]+]]:fpr64 = LDRDui [[UBFMXri]], 1
-  ; CHECK-NEXT:   [[ADDSXri5:%[0-9]+]]:gpr64 = ADDSXri [[COPY7]], 1, 0, implicit-def $nzcv
-  ; CHECK-NEXT:   [[UBFMXri1:%[0-9]+]]:gpr64common = UBFMXri [[COPY7]], 61, 60
-  ; CHECK-NEXT:   [[COPY8:%[0-9]+]]:gpr64all = COPY [[ADDSXri5]]
+  ; CHECK-NEXT:   [[ADDSXri5:%[0-9]+]]:gpr64 = ADDSXri [[COPY6]], 1, 0, implicit-def $nzcv
+  ; CHECK-NEXT:   [[UBFMXri1:%[0-9]+]]:gpr64common = UBFMXri [[COPY6]], 61, 60
+  ; CHECK-NEXT:   [[COPY7:%[0-9]+]]:gpr64all = COPY [[ADDSXri5]]
   ; CHECK-NEXT: {{  $}}
   ; CHECK-NEXT: bb.9:
   ; CHECK-NEXT:   successors: %bb.9(0x40000000), %bb.10(0x40000000)
   ; CHECK-NEXT: {{  $}}
   ; CHECK-NEXT:   [[PHI6:%[0-9]+]]:gpr64 = PHI %38, %bb.9, [[ADDSXri4]], %bb.8
   ; CHECK-NEXT:   [[PHI7:%[0-9]+]]:gpr64common = PHI %40, %bb.9, [[UBFMXri]], %bb.8
-  ; CHECK-NEXT:   [[PHI8:%[0-9]+]]:gpr64all = PHI %43, %bb.9, [[COPY7]], %bb.8
+  ; CHECK-NEXT:   [[PHI8:%[0-9]+]]:gpr64all = PHI %43, %bb.9, [[COPY6]], %bb.8
   ; CHECK-NEXT:   [[PHI9:%[0-9]+]]:fpr64 = PHI %45, %bb.9, [[LDRDui2]], %bb.8
   ; CHECK-NEXT:   [[PHI10:%[0-9]+]]:gpr64 = PHI %47, %bb.9, [[ADDSXri5]], %bb.8
   ; CHECK-NEXT:   [[PHI11:%[0-9]+]]:gpr64common = PHI %49, %bb.9, [[UBFMXri1]], %bb.8
   ; CHECK-NEXT:   [[PHI12:%[0-9]+]]:fpr64 = PHI %51, %bb.9, [[PHI5]], %bb.8
-  ; CHECK-NEXT:   [[PHI13:%[0-9]+]]:gpr64common = PHI %53, %bb.9, [[COPY8]], %bb.8
+  ; CHECK-NEXT:   [[PHI13:%[0-9]+]]:gpr64common = PHI %53, %bb.9, [[COPY7]], %bb.8
   ; CHECK-NEXT:   [[LDRDui3:%[0-9]+]]:fpr64 = LDRDui [[PHI11]], 1
   ; CHECK-NEXT:   [[ADDSXri6:%[0-9]+]]:gpr64 = ADDSXri [[PHI13]], 1, 0, implicit-def $nzcv
   ; CHECK-NEXT:   [[UBFMXri2:%[0-9]+]]:gpr64common = UBFMXri [[PHI13]], 61, 60
   ; CHECK-NEXT:   [[FADDDrr:%[0-9]+]]:fpr64 = nofpexcept FADDDrr [[PHI12]], [[PHI9]], implicit $fpcr
-  ; CHECK-NEXT:   [[COPY9:%[0-9]+]]:gpr64common = COPY [[ADDSXri6]]
+  ; CHECK-NEXT:   [[COPY8:%[0-9]+]]:gpr64common = COPY [[ADDSXri6]]
   ; CHECK-NEXT:   [[LDRDui4:%[0-9]+]]:fpr64 = LDRDui [[UBFMXri2]], 1
-  ; CHECK-NEXT:   [[ADDSXri7:%[0-9]+]]:gpr64common = ADDSXri [[COPY9]], 1, 0, implicit-def $nzcv
-  ; CHECK-NEXT:   [[UBFMXri3:%[0-9]+]]:gpr64common = UBFMXri [[COPY9]], 61, 60
+  ; CHECK-NEXT:   [[ADDSXri7:%[0-9]+]]:gpr64common = ADDSXri [[COPY8]], 1, 0, implicit-def $nzcv
+  ; CHECK-NEXT:   [[UBFMXri3:%[0-9]+]]:gpr64common = UBFMXri [[COPY8]], 61, 60
   ; CHECK-NEXT:   [[FADDDrr1:%[0-9]+]]:fpr64 = nofpexcept FADDDrr [[FADDDrr]], [[LDRDui3]], implicit $fpcr
-  ; CHECK-NEXT:   [[COPY10:%[0-9]+]]:gpr64all = COPY [[ADDSXri7]]
+  ; CHECK-NEXT:   [[COPY9:%[0-9]+]]:gpr64all = COPY [[ADDSXri7]]
   ; CHECK-NEXT:   [[CSINCXr4:%[0-9]+]]:gpr64common = CSINCXr $xzr, $xzr, 3, implicit $nzcv
   ; CHECK-NEXT:   [[ADDSXri8:%[0-9]+]]:gpr64 = ADDSXri [[ADDSXri7]], 1, 0, implicit-def $nzcv
   ; CHECK-NEXT:   [[CSINCXr5:%[0-9]+]]:gpr64common = CSINCXr [[CSINCXr4]], [[CSINCXr4]], 3, implicit $nzcv
@@ -131,7 +130,7 @@ body:             |
   ; CHECK-NEXT:   [[LDRDui5:%[0-9]+]]:fpr64 = LDRDui [[UBFMXri3]], 1
   ; CHECK-NEXT:   [[FADDDrr2:%[0-9]+]]:fpr64 = nofpexcept FADDDrr [[FADDDrr1]], [[LDRDui4]], implicit $fpcr
   ; CHECK-NEXT:   [[FADDDrr3:%[0-9]+]]:fpr64 = nofpexcept FADDDrr [[FADDDrr2]], [[LDRDui5]], implicit $fpcr
-  ; CHECK-NEXT:   [[ADDSXri9:%[0-9]+]]:gpr64 = ADDSXri [[COPY9]], 1, 0, implicit-def $nzcv
+  ; CHECK-NEXT:   [[ADDSXri9:%[0-9]+]]:gpr64 = ADDSXri [[COPY8]], 1, 0, implicit-def $nzcv
   ; CHECK-NEXT:   [[CSINCXr6:%[0-9]+]]:gpr64common = CSINCXr $xzr, $xzr, 3, implicit $nzcv
   ; CHECK-NEXT:   dead $xzr = SUBSXri [[CSINCXr6]], 0, 0, implicit-def $nzcv
   ; CHECK-NEXT:   Bcc 0, %bb.11, implicit $nzcv
@@ -140,7 +139,7 @@ body:             |
   ; CHECK-NEXT: bb.11:
   ; CHECK-NEXT:   successors: %bb.6(0x80000000)
   ; CHECK-NEXT: {{  $}}
-  ; CHECK-NEXT:   [[PHI14:%[0-9]+]]:gpr64common = PHI [[COPY6]], %bb.7, [[COPY10]], %bb.10
+  ; CHECK-NEXT:   [[PHI14:%[0-9]+]]:gpr64common = PHI [[COPY5]], %bb.7, [[COPY9]], %bb.10
   ; CHECK-NEXT:   [[PHI15:%[0-9]+]]:fpr64 = PHI [[PHI5]], %bb.7, [[FADDDrr3]], %bb.10
   ; CHECK-NEXT:   B %bb.6
   ; CHECK-NEXT: {{  $}}
@@ -159,7 +158,7 @@ body:             |
   ; CHECK-NEXT:   [[LDRDui6:%[0-9]+]]:fpr64 = LDRDui [[UBFMXri4]], 1
   ; CHECK-NEXT:   [[FADDDrr4:%[0-9]+]]:fpr64 = nofpexcept FADDDrr [[PHI18]], [[LDRDui6]], implicit $fpcr
   ; CHECK-NEXT:   [[ADDSXri10:%[0-9]+]]:gpr64 = ADDSXri [[PHI17]], 1, 0, implicit-def $nzcv
-  ; CHECK-NEXT:   [[COPY11:%[0-9]+]]:gpr64all = COPY [[ADDSXri10]]
+  ; CHECK-NEXT:   [[COPY10:%[0-9]+]]:gpr64all = COPY [[ADDSXri10]]
   ; CHECK-NEXT:   Bcc 2, %bb.12, implicit $nzcv
   ; CHECK-NEXT:   B %bb.6
   bb.0:
diff --git a/llvm/test/CodeGen/AArch64/sms-subs-nzcv-private.mir b/llvm/test/CodeGen/AArch64/sms-subs-nzcv-private.mir
new file mode 100644
index 0000000000000..3c5dc1b23b2f7
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sms-subs-nzcv-private.mir
@@ -0,0 +1,126 @@
+# RUN: llc --verify-machineinstrs -mtriple=aarch64-apple-macosx -mcpu=apple-m1 \
+# RUN:   -run-pass pipeliner -debug-only=pipeliner -o /dev/null %s 2>&1 | \
+# RUN:   FileCheck %s
+# RUN: llc --verify-machineinstrs -mtriple=aarch64-apple-macosx -mcpu=apple-m1 \
+# RUN:   -run-pass pipeliner -debug-only=pipeliner -window-sched=force -o /dev/null %s 2>&1 | \
+# RUN:   FileCheck %s
+
+# REQUIRES: asserts
+
+# Check that we allow pipelining of loops with a phys reg def of NZCV, but only
+# when one use is private to the condition on the loop backedge. This allows
+# pipelining on a common class of loops on aarch64 where the exit condition is
+# predicated on the decrement of the trip count.
+
+# CHECK:      Special MI defined by target is not allowed in window scheduling:
+# CHECK-NEXT: %{{.*}}:gpr64 = nsw SUBSXri %{{.*}}:gpr64sp, 1, 0, implicit-def $nzcv
+# CHECK-NOT:  Special MI defined by target is not allowed in window scheduling:
+
+--- |
+  define void @private_nzcv(i64 %n, float %fa, ptr noalias %x, ptr noalias %y) {
+  entry:
+    br i1 false, label %preheader, label %exit
+  preheader:
+    br label %for.body
+  for.body:
+    br i1 false, label %exit, label %for.body
+  exit:
+    ret void
+  }
+  define void @shared_nzcv(i64 %n, float %fa, ptr %fx) {
+  entry:
+    br i1 false, label %preheader, label %exit
+  preheader:
+    br label %for.body
+  for.body:
+    br i1 false, label %exit, label %for.body
+  exit:
+    ret void
+  }
+...
+---
+name:            private_nzcv
+tracksRegLiveness: true
+liveins:
+  - { reg: '$x0', virtual-reg: '%3' }
+  - { reg: '$s0', virtual-reg: '%4' }
+  - { reg: '$x1', virtual-reg: '%5' }
+  - { reg: '$x2', virtual-reg: '%6' }
+body:             |
+  bb.0.entry:
+    successors: %bb.1(0x50000000), %bb.2(0x30000000)
+    liveins: $x0, $s0, $x1, $x2
+
+    %3:gpr64sp = COPY $x0
+    %4:fpr32 = COPY $s0
+    %5:gpr64sp = COPY $x1
+    %6:gpr64sp = COPY $x2
+    dead $xzr = SUBSXri %3, 1, 0, implicit-def $nzcv
+    Bcc 3, %bb.2, implicit $nzcv
+    B %bb.1
+
+  bb.1.preheader:
+    %0:gpr64sp = COPY %3
+    B %bb.3
+
+  bb.2.exit:
+    RET_ReallyLR
+
+  bb.3.for.body:
+    successors: %bb.2(0x04000000), %bb.3(0x7c000000)
+
+    %1:gpr64sp = PHI %0, %bb.1, %7, %bb.3
+    %8:gpr64sp = PHI %5, %bb.1, %9, %bb.3
+    %10:gpr64sp = PHI %6, %bb.1, %11, %bb.3
+    early-clobber %9:gpr64sp, %12:fpr32 = LDRSpost %8, 4 :: (load (s32) from %ir.x)
+    %13:fpr32 = nofpexcept FMULSrr %4, killed %12, implicit $fpcr
+    %14:fpr32 = nofpexcept FMULSrr %4, killed %13, implicit $fpcr
+    early-clobber %11:gpr64sp = STRSpost killed %14, %10, 4 :: (store (s32) into %ir.y)
+    %15:gpr64 = nsw SUBSXri %1, 1, 0, implicit-def $nzcv
+    %7:gpr64all = COPY %15
+    Bcc 1, %bb.3, implicit $nzcv
+    B %bb.2
+
+...
+---
+name:            shared_nzcv
+tracksRegLiveness: true
+liveins:
+  - { reg: '$x0', virtual-reg: '%3' }
+  - { reg: '$s0', virtual-reg: '%4' }
+  - { reg: '$x1', virtual-reg: '%5' }
+body:             |
+  bb.0.entry:
+    successors: %bb.1(0x50000000), %bb.2(0x30000000)
+    liveins: $x0, $s0, $x1
+
+    %3:gpr64sp = COPY $x0
+    %4:fpr32 = COPY $s0
+    %5:gpr64sp = COPY $x1
+    dead $xzr = SUBSXri %3, 1, 0, implicit-def $nzcv
+    Bcc 3, %bb.2, implicit $nzcv
+    B %bb.1
+
+  bb.1.preheader:
+    %0:gpr64sp = COPY %3
+    B %bb.3
+
+  bb.2.exit:
+    RET_ReallyLR
+
+  bb.3.for.body:
+    successors: %bb.2(0x04000000), %bb.3(0x7c000000)
+
+    %1:gpr64sp = PHI %0, %bb.1, %6, %bb.3
+    %9:gpr64sp = PHI %5, %bb.1, %10, %bb.3
+    %11:fpr32 = LDRSui %9, 0 :: (load (s32))
+    %15:gpr64 = nsw SUBSXri %1, 1, 0, implicit-def $nzcv
+    %16:fpr32 = FCSELSrrr %4, killed %11, 1, implicit $nzcv
+    %12:fpr32 = nofpexcept FMULSrr %4, killed %16, implicit $fpcr
+    %14:fpr32 = nofpexcept FMULSrr %4, killed %12, implicit $fpcr
+    early-clobber %10:gpr64sp = STRSpost killed %14, killed %9, 4 :: (store (s32))
+    %6:gpr64all = COPY %15
+    Bcc 1, %bb.3, implicit $nzcv
+    B %bb.2
+
+...



More information about the llvm-commits mailing list