[llvm] [llvm][CodeGen][AArch64] Allow the WindowScheduler to pipeline SUBS+Bcc-terminated loops (PR #191587)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Apr 10 19:10:47 PDT 2026
llvmbot wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-aarch64
Author: Jon Roelofs (jroelofs)
<details>
<summary>Changes</summary>
Previously they were ignored because SUBS has an implicit phys reg def, but we can relax that a bit when NZCV is known to only be used by the loop terminator as the dependence will naturally constrain the SUBS to stage 0 along with the branch.
---
Full diff: https://github.com/llvm/llvm-project/pull/191587.diff
7 Files Affected:
- (modified) llvm/include/llvm/CodeGen/TargetInstrInfo.h (+14-2)
- (modified) llvm/lib/CodeGen/WindowScheduler.cpp (+2-7)
- (modified) llvm/lib/Target/AArch64/AArch64InstrInfo.cpp (+21-4)
- (modified) llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp (+3-2)
- (modified) llvm/lib/Target/PowerPC/PPCInstrInfo.cpp (+3-2)
- (modified) llvm/lib/Target/RISCV/RISCVInstrInfo.cpp (+1-1)
- (added) llvm/test/CodeGen/AArch64/sms-subs-nzcv-private.mir (+126)
``````````diff
diff --git a/llvm/include/llvm/CodeGen/TargetInstrInfo.h b/llvm/include/llvm/CodeGen/TargetInstrInfo.h
index cd5561e57d033..c2d4fc7208060 100644
--- a/llvm/include/llvm/CodeGen/TargetInstrInfo.h
+++ b/llvm/include/llvm/CodeGen/TargetInstrInfo.h
@@ -822,8 +822,20 @@ class LLVM_ABI TargetInstrInfo : public MCInstrInfo {
virtual ~PipelinerLoopInfo();
/// Return true if the given instruction should not be pipelined and should
/// be ignored. An example could be a loop comparison, or induction variable
- /// update with no users being pipelined.
- virtual bool shouldIgnoreForPipelining(const MachineInstr *MI) const = 0;
+ /// update with no users being pipelined. By default we ignore instructions
+ /// with physical register defs because the pipeliner cannot reason about
+ /// physical register lifetimes.
+ virtual bool shouldIgnoreForPipelining(const MachineInstr *MI) const {
+ for (const auto &MO : MI->all_defs())
+ if (MO.isReg() && MO.getReg().isPhysical()) {
+ DEBUG_WITH_TYPE(
+ "pipeliner",
+ dbgs() << "MI defines a physical reg; ignoring for pipelining:\n"
+ << *MI << "\n");
+ return true;
+ }
+ return false;
+ }
/// Return true if the proposed schedule should used. Otherwise return
/// false to not pipeline the loop. This function should be used to ensure
diff --git a/llvm/lib/CodeGen/WindowScheduler.cpp b/llvm/lib/CodeGen/WindowScheduler.cpp
index 2492dfc3ca553..aaac27dff65b8 100644
--- a/llvm/lib/CodeGen/WindowScheduler.cpp
+++ b/llvm/lib/CodeGen/WindowScheduler.cpp
@@ -229,15 +229,10 @@ bool WindowScheduler::initialize() {
}
if (PLI->shouldIgnoreForPipelining(&MI)) {
LLVM_DEBUG(dbgs() << "Special MI defined by target is not allowed in "
- "window scheduling!\n");
+ "window scheduling:\n"
+ << MI << "\n");
return false;
}
- for (auto &Def : MI.all_defs())
- if (Def.isReg() && Def.getReg().isPhysical()) {
- LLVM_DEBUG(dbgs() << "Physical registers are not supported in "
- "window scheduling!\n");
- return false;
- }
}
if (SchedInstrNum <= WindowRegionLimit) {
LLVM_DEBUG(dbgs() << "There are too few MIs in the window region!\n");
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
index 4094526574d7a..3235a2f3881ad 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.cpp
@@ -11410,6 +11410,10 @@ class AArch64PipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
/// The normalized condition used by createTripCountGreaterCondition()
SmallVector<MachineOperand, 4> Cond;
+ /// True iff \p CondBranch is the only use of \p Comp's NZCV def, making it
+ /// private to the back-edge of the loop.
+ bool NZCVIsBackedgePrivate;
+
public:
AArch64PipelinerLoopInfo(MachineBasicBlock *LoopBB, MachineInstr *CondBranch,
MachineInstr *Comp, unsigned CompCounterOprNum,
@@ -11422,12 +11426,25 @@ class AArch64PipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
LoopBB(LoopBB), CondBranch(CondBranch), Comp(Comp),
CompCounterOprNum(CompCounterOprNum), Update(Update),
UpdateCounterOprNum(UpdateCounterOprNum), Init(Init),
- IsUpdatePriorComp(IsUpdatePriorComp), Cond(Cond.begin(), Cond.end()) {}
+ IsUpdatePriorComp(IsUpdatePriorComp), Cond(Cond.begin(), Cond.end()),
+ NZCVIsBackedgePrivate(true) {
+ for (const MachineOperand &MO : MRI.use_nodbg_operands(AArch64::NZCV)) {
+ if (MO.getParent()->getParent() == LoopBB &&
+ MO.getParent() != CondBranch) {
+ NZCVIsBackedgePrivate = false;
+ break;
+ }
+ }
+ }
bool shouldIgnoreForPipelining(const MachineInstr *MI) const override {
- // Make the instructions for loop control be placed in stage 0.
- // The predecessors of Comp are considered by the caller.
- return MI == Comp;
+ if (MI == Comp) {
+ // If Comp's NZCV is only used by CondBranch, the loop-carried counter
+ // dependency naturally constrains Comp to stage 0, and no explicit
+ // pinning is needed.
+ return !NZCVIsBackedgePrivate;
+ }
+ return PipelinerLoopInfo::shouldIgnoreForPipelining(MI);
}
std::optional<bool> createTripCountGreaterCondition(
diff --git a/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp b/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
index 732256f61556f..cd24c32221f92 100644
--- a/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
+++ b/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
@@ -748,8 +748,9 @@ class HexagonPipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
}
bool shouldIgnoreForPipelining(const MachineInstr *MI) const override {
- // Only ignore the terminator.
- return MI == EndLoop;
+ if (MI == EndLoop)
+ return true;
+ return PipelinerLoopInfo::shouldIgnoreForPipelining(MI);
}
std::optional<bool> createTripCountGreaterCondition(
diff --git a/llvm/lib/Target/PowerPC/PPCInstrInfo.cpp b/llvm/lib/Target/PowerPC/PPCInstrInfo.cpp
index 6d95314d4b019..fec09ca6d8017 100644
--- a/llvm/lib/Target/PowerPC/PPCInstrInfo.cpp
+++ b/llvm/lib/Target/PowerPC/PPCInstrInfo.cpp
@@ -5753,8 +5753,9 @@ class PPCPipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
}
bool shouldIgnoreForPipelining(const MachineInstr *MI) const override {
- // Only ignore the terminator.
- return MI == EndLoop;
+ if (MI == EndLoop)
+ return true;
+ return PipelinerLoopInfo::shouldIgnoreForPipelining(MI);
}
std::optional<bool> createTripCountGreaterCondition(
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfo.cpp b/llvm/lib/Target/RISCV/RISCVInstrInfo.cpp
index db559f4949904..880b2ea909fb5 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfo.cpp
@@ -5251,7 +5251,7 @@ class RISCVPipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
return true;
if (RHS && MI == RHS)
return true;
- return false;
+ return PipelinerLoopInfo::shouldIgnoreForPipelining(MI);
}
std::optional<bool> createTripCountGreaterCondition(
diff --git a/llvm/test/CodeGen/AArch64/sms-subs-nzcv-private.mir b/llvm/test/CodeGen/AArch64/sms-subs-nzcv-private.mir
new file mode 100644
index 0000000000000..3c5dc1b23b2f7
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/sms-subs-nzcv-private.mir
@@ -0,0 +1,126 @@
+# RUN: llc --verify-machineinstrs -mtriple=aarch64-apple-macosx -mcpu=apple-m1 \
+# RUN: -run-pass pipeliner -debug-only=pipeliner -o /dev/null %s 2>&1 | \
+# RUN: FileCheck %s
+# RUN: llc --verify-machineinstrs -mtriple=aarch64-apple-macosx -mcpu=apple-m1 \
+# RUN: -run-pass pipeliner -debug-only=pipeliner -window-sched=force -o /dev/null %s 2>&1 | \
+# RUN: FileCheck %s
+
+# REQUIRES: asserts
+
+# Check that we allow pipelining of loops with a phys reg def of NZCV, but only
+# when one use is private to the condition on the loop backedge. This allows
+# pipelining on a common class of loops on aarch64 where the exit condition is
+# predicated on the decrement of the trip count.
+
+# CHECK: Special MI defined by target is not allowed in window scheduling:
+# CHECK-NEXT: %{{.*}}:gpr64 = nsw SUBSXri %{{.*}}:gpr64sp, 1, 0, implicit-def $nzcv
+# CHECK-NOT: Special MI defined by target is not allowed in window scheduling:
+
+--- |
+ define void @private_nzcv(i64 %n, float %fa, ptr noalias %x, ptr noalias %y) {
+ entry:
+ br i1 false, label %preheader, label %exit
+ preheader:
+ br label %for.body
+ for.body:
+ br i1 false, label %exit, label %for.body
+ exit:
+ ret void
+ }
+ define void @shared_nzcv(i64 %n, float %fa, ptr %fx) {
+ entry:
+ br i1 false, label %preheader, label %exit
+ preheader:
+ br label %for.body
+ for.body:
+ br i1 false, label %exit, label %for.body
+ exit:
+ ret void
+ }
+...
+---
+name: private_nzcv
+tracksRegLiveness: true
+liveins:
+ - { reg: '$x0', virtual-reg: '%3' }
+ - { reg: '$s0', virtual-reg: '%4' }
+ - { reg: '$x1', virtual-reg: '%5' }
+ - { reg: '$x2', virtual-reg: '%6' }
+body: |
+ bb.0.entry:
+ successors: %bb.1(0x50000000), %bb.2(0x30000000)
+ liveins: $x0, $s0, $x1, $x2
+
+ %3:gpr64sp = COPY $x0
+ %4:fpr32 = COPY $s0
+ %5:gpr64sp = COPY $x1
+ %6:gpr64sp = COPY $x2
+ dead $xzr = SUBSXri %3, 1, 0, implicit-def $nzcv
+ Bcc 3, %bb.2, implicit $nzcv
+ B %bb.1
+
+ bb.1.preheader:
+ %0:gpr64sp = COPY %3
+ B %bb.3
+
+ bb.2.exit:
+ RET_ReallyLR
+
+ bb.3.for.body:
+ successors: %bb.2(0x04000000), %bb.3(0x7c000000)
+
+ %1:gpr64sp = PHI %0, %bb.1, %7, %bb.3
+ %8:gpr64sp = PHI %5, %bb.1, %9, %bb.3
+ %10:gpr64sp = PHI %6, %bb.1, %11, %bb.3
+ early-clobber %9:gpr64sp, %12:fpr32 = LDRSpost %8, 4 :: (load (s32) from %ir.x)
+ %13:fpr32 = nofpexcept FMULSrr %4, killed %12, implicit $fpcr
+ %14:fpr32 = nofpexcept FMULSrr %4, killed %13, implicit $fpcr
+ early-clobber %11:gpr64sp = STRSpost killed %14, %10, 4 :: (store (s32) into %ir.y)
+ %15:gpr64 = nsw SUBSXri %1, 1, 0, implicit-def $nzcv
+ %7:gpr64all = COPY %15
+ Bcc 1, %bb.3, implicit $nzcv
+ B %bb.2
+
+...
+---
+name: shared_nzcv
+tracksRegLiveness: true
+liveins:
+ - { reg: '$x0', virtual-reg: '%3' }
+ - { reg: '$s0', virtual-reg: '%4' }
+ - { reg: '$x1', virtual-reg: '%5' }
+body: |
+ bb.0.entry:
+ successors: %bb.1(0x50000000), %bb.2(0x30000000)
+ liveins: $x0, $s0, $x1
+
+ %3:gpr64sp = COPY $x0
+ %4:fpr32 = COPY $s0
+ %5:gpr64sp = COPY $x1
+ dead $xzr = SUBSXri %3, 1, 0, implicit-def $nzcv
+ Bcc 3, %bb.2, implicit $nzcv
+ B %bb.1
+
+ bb.1.preheader:
+ %0:gpr64sp = COPY %3
+ B %bb.3
+
+ bb.2.exit:
+ RET_ReallyLR
+
+ bb.3.for.body:
+ successors: %bb.2(0x04000000), %bb.3(0x7c000000)
+
+ %1:gpr64sp = PHI %0, %bb.1, %6, %bb.3
+ %9:gpr64sp = PHI %5, %bb.1, %10, %bb.3
+ %11:fpr32 = LDRSui %9, 0 :: (load (s32))
+ %15:gpr64 = nsw SUBSXri %1, 1, 0, implicit-def $nzcv
+ %16:fpr32 = FCSELSrrr %4, killed %11, 1, implicit $nzcv
+ %12:fpr32 = nofpexcept FMULSrr %4, killed %16, implicit $fpcr
+ %14:fpr32 = nofpexcept FMULSrr %4, killed %12, implicit $fpcr
+ early-clobber %10:gpr64sp = STRSpost killed %14, killed %9, 4 :: (store (s32))
+ %6:gpr64all = COPY %15
+ Bcc 1, %bb.3, implicit $nzcv
+ B %bb.2
+
+...
``````````
</details>
https://github.com/llvm/llvm-project/pull/191587
More information about the llvm-commits
mailing list