[llvm] 507d823 - [LSR] Use TTI to check if zero-start IV is free in getSetupCost (#190587)
via llvm-commits
llvm-commits at lists.llvm.org
Sun Apr 12 17:18:32 PDT 2026
Author: Wenju He
Date: 2026-04-13T08:18:28+08:00
New Revision: 507d82339230990d65ccfe753d0fd3e620c82b71
URL: https://github.com/llvm/llvm-project/commit/507d82339230990d65ccfe753d0fd3e620c82b71
DIFF: https://github.com/llvm/llvm-project/commit/507d82339230990d65ccfe753d0fd3e620c82b71.diff
LOG: [LSR] Use TTI to check if zero-start IV is free in getSetupCost (#190587)
This avoids a downstream regression where LSR prefers {-1,+1}.
When constant zero typically doesn't require preheader initialization
(queried via TTI::getIntImmCost), consider it as free in getSetupCost.
Three test changes are improvements: amx-across-func.ll,
2011-11-29-postincphi.ll and pr62660-normalization-failure.ll.
Other test changes are neutral.
Added:
Modified:
llvm/lib/Transforms/Scalar/LoopStrengthReduce.cpp
llvm/test/CodeGen/X86/AMX/amx-across-func.ll
llvm/test/Transforms/LoopStrengthReduce/RISCV/lsr-drop-solution-dbg-msg.ll
llvm/test/Transforms/LoopStrengthReduce/X86/2011-11-29-postincphi.ll
llvm/test/Transforms/LoopStrengthReduce/X86/postinc-iv-used-by-urem-and-udiv.ll
llvm/test/Transforms/LoopStrengthReduce/X86/pr62660-normalization-failure.ll
llvm/test/Transforms/LoopStrengthReduce/duplicated-phis.ll
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Scalar/LoopStrengthReduce.cpp b/llvm/lib/Transforms/Scalar/LoopStrengthReduce.cpp
index 5421cad31c3ba..01322c26aa77e 100644
--- a/llvm/lib/Transforms/Scalar/LoopStrengthReduce.cpp
+++ b/llvm/lib/Transforms/Scalar/LoopStrengthReduce.cpp
@@ -1366,23 +1366,31 @@ static bool isAMCompletelyFolded(const TargetTransformInfo &TTI,
bool HasBaseReg, int64_t Scale,
Instruction *Fixup = nullptr);
-static unsigned getSetupCost(const SCEV *Reg, unsigned Depth) {
- if (isa<SCEVUnknown>(Reg) || isa<SCEVConstant>(Reg))
+static unsigned getSetupCost(const SCEV *Reg, unsigned Depth,
+ const TargetTransformInfo &TTI) {
+ if (isa<SCEVUnknown>(Reg))
return 1;
+ if (const auto *C = dyn_cast<SCEVConstant>(Reg)) {
+ if (TTI.getIntImmCost(C->getAPInt(), C->getType(),
+ TargetTransformInfo::TCK_RecipThroughput) ==
+ TargetTransformInfo::TCC_Free)
+ return 0;
+ return 1;
+ }
if (Depth == 0)
return 0;
if (const auto *S = dyn_cast<SCEVAddRecExpr>(Reg))
- return getSetupCost(S->getStart(), Depth - 1);
+ return getSetupCost(S->getStart(), Depth - 1, TTI);
if (auto S = dyn_cast<SCEVIntegralCastExpr>(Reg))
- return getSetupCost(S->getOperand(), Depth - 1);
+ return getSetupCost(S->getOperand(), Depth - 1, TTI);
if (auto S = dyn_cast<SCEVNAryExpr>(Reg))
return std::accumulate(S->operands().begin(), S->operands().end(), 0,
[&](unsigned i, const SCEV *Reg) {
- return i + getSetupCost(Reg, Depth - 1);
+ return i + getSetupCost(Reg, Depth - 1, TTI);
});
if (auto S = dyn_cast<SCEVUDivExpr>(Reg))
- return getSetupCost(S->getLHS(), Depth - 1) +
- getSetupCost(S->getRHS(), Depth - 1);
+ return getSetupCost(S->getLHS(), Depth - 1, TTI) +
+ getSetupCost(S->getRHS(), Depth - 1, TTI);
return 0;
}
@@ -1452,7 +1460,7 @@ void Cost::RateRegister(const Formula &F, const SCEV *Reg,
// Rough heuristic; favor registers which don't require extra setup
// instructions in the preheader.
- C.SetupCost += getSetupCost(Reg, SetupCostDepthLimit);
+ C.SetupCost += getSetupCost(Reg, SetupCostDepthLimit, *TTI);
// Ensure we don't, even with the recusion limit, produce invalid costs.
C.SetupCost = std::min<unsigned>(C.SetupCost, 1 << 16);
diff --git a/llvm/test/CodeGen/X86/AMX/amx-across-func.ll b/llvm/test/CodeGen/X86/AMX/amx-across-func.ll
index 2bda8db040296..1e752ef981960 100644
--- a/llvm/test/CodeGen/X86/AMX/amx-across-func.ll
+++ b/llvm/test/CodeGen/X86/AMX/amx-across-func.ll
@@ -230,7 +230,7 @@ define dso_local i32 @test_loop(i32 %0) nounwind {
; CHECK-NEXT: testl %ebx, %ebx
; CHECK-NEXT: jg .LBB2_4
; CHECK-NEXT: # %bb.1: # %.preheader
-; CHECK-NEXT: movl $7, %ebp
+; CHECK-NEXT: xorl %ebp, %ebp
; CHECK-NEXT: movl $buf, %r14d
; CHECK-NEXT: movl $32, %r15d
; CHECK-NEXT: movw $8, %r12w
@@ -248,13 +248,12 @@ define dso_local i32 @test_loop(i32 %0) nounwind {
; CHECK-NEXT: callq foo
; CHECK-NEXT: ldtilecfg (%rsp)
; CHECK-NEXT: decl %ebp
-; CHECK-NEXT: cmpl $7, %ebp
; CHECK-NEXT: jne .LBB2_2
; CHECK-NEXT: # %bb.3:
; CHECK-NEXT: cmpl $3, %ebx
; CHECK-NEXT: jne .LBB2_4
; CHECK-NEXT: # %bb.6:
-; CHECK-NEXT: testl %ebp, %ebp
+; CHECK-NEXT: cmpl $-7, %ebp
; CHECK-NEXT: jne .LBB2_5
; CHECK-NEXT: # %bb.7:
; CHECK-NEXT: incl %ebx
@@ -295,7 +294,7 @@ define dso_local i32 @test_loop(i32 %0) nounwind {
; IPRA-NEXT: testl %edi, %edi
; IPRA-NEXT: jg .LBB2_4
; IPRA-NEXT: # %bb.1: # %.preheader
-; IPRA-NEXT: movl $7, %ecx
+; IPRA-NEXT: xorl %ecx, %ecx
; IPRA-NEXT: movl $buf, %edx
; IPRA-NEXT: movl $32, %esi
; IPRA-NEXT: movw $8, %di
@@ -307,13 +306,12 @@ define dso_local i32 @test_loop(i32 %0) nounwind {
; IPRA-NEXT: tilestored %tmm0, (%r8,%rsi)
; IPRA-NEXT: callq foo
; IPRA-NEXT: decl %ecx
-; IPRA-NEXT: cmpl $7, %ecx
; IPRA-NEXT: jne .LBB2_2
; IPRA-NEXT: # %bb.3:
; IPRA-NEXT: cmpl $3, %eax
; IPRA-NEXT: jne .LBB2_4
; IPRA-NEXT: # %bb.6:
-; IPRA-NEXT: testl %ecx, %ecx
+; IPRA-NEXT: cmpl $-7, %ecx
; IPRA-NEXT: jne .LBB2_5
; IPRA-NEXT: # %bb.7:
; IPRA-NEXT: incl %eax
diff --git a/llvm/test/Transforms/LoopStrengthReduce/RISCV/lsr-drop-solution-dbg-msg.ll b/llvm/test/Transforms/LoopStrengthReduce/RISCV/lsr-drop-solution-dbg-msg.ll
index 8d9d43202f0d9..3eade6474e61f 100644
--- a/llvm/test/Transforms/LoopStrengthReduce/RISCV/lsr-drop-solution-dbg-msg.ll
+++ b/llvm/test/Transforms/LoopStrengthReduce/RISCV/lsr-drop-solution-dbg-msg.ll
@@ -7,7 +7,7 @@ target triple = "riscv64-unknown-linux-gnu"
define ptr @foo(ptr %a0, ptr %a1, i64 %a2) {
;DEBUG: The baseline solution requires 2 instructions 4 regs, with addrec cost 2, plus 3 setup cost
-;DEBUG: The chosen solution requires 3 instructions 6 regs, with addrec cost 1, plus 2 base adds, plus 5 setup cost
+;DEBUG: The chosen solution requires 3 instructions 6 regs, with addrec cost 1, plus 2 base adds, plus 4 setup cost
;DEBUG: Baseline is more profitable than chosen solution, dropping LSR solution.
;DEBUG2: Baseline is more profitable than chosen solution, add option 'lsr-drop-solution' to drop LSR solution.
diff --git a/llvm/test/Transforms/LoopStrengthReduce/X86/2011-11-29-postincphi.ll b/llvm/test/Transforms/LoopStrengthReduce/X86/2011-11-29-postincphi.ll
index 7ae78ae6a1fd4..565b1c6c71195 100644
--- a/llvm/test/Transforms/LoopStrengthReduce/X86/2011-11-29-postincphi.ll
+++ b/llvm/test/Transforms/LoopStrengthReduce/X86/2011-11-29-postincphi.ll
@@ -15,21 +15,19 @@ define i64 @sqlite3DropTriggerPtr() nounwind {
; CHECK-LABEL: sqlite3DropTriggerPtr:
; CHECK: # %bb.0: # %bb
; CHECK-NEXT: pushq %rbx
-; CHECK-NEXT: movl $1, %ebx
+; CHECK-NEXT: xorl %ebx, %ebx
; CHECK-NEXT: callq check at PLT
; CHECK-NEXT: .p2align 4
; CHECK-NEXT: .LBB0_1: # %bb1
; CHECK-NEXT: # =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: incq %rbx
; CHECK-NEXT: testb %al, %al
-; CHECK-NEXT: je .LBB0_4
+; CHECK-NEXT: je .LBB0_3
; CHECK-NEXT: # %bb.2: # %bb4
; CHECK-NEXT: # in Loop: Header=BB0_1 Depth=1
-; CHECK-NEXT: incq %rbx
; CHECK-NEXT: testb %al, %al
; CHECK-NEXT: jne .LBB0_1
-; CHECK-NEXT: # %bb.3: # %bb8split
-; CHECK-NEXT: decq %rbx
-; CHECK-NEXT: .LBB0_4: # %bb8
+; CHECK-NEXT: .LBB0_3: # %bb8
; CHECK-NEXT: movq %rbx, %rax
; CHECK-NEXT: popq %rbx
; CHECK-NEXT: retq
diff --git a/llvm/test/Transforms/LoopStrengthReduce/X86/postinc-iv-used-by-urem-and-udiv.ll b/llvm/test/Transforms/LoopStrengthReduce/X86/postinc-iv-used-by-urem-and-udiv.ll
index 838b48aa56906..1fec73972ec51 100644
--- a/llvm/test/Transforms/LoopStrengthReduce/X86/postinc-iv-used-by-urem-and-udiv.ll
+++ b/llvm/test/Transforms/LoopStrengthReduce/X86/postinc-iv-used-by-urem-and-udiv.ll
@@ -93,19 +93,19 @@ define i32 @test_pr62852() {
; CHECK-NEXT: entry:
; CHECK-NEXT: br label [[LOOP:%.*]]
; CHECK: loop:
-; CHECK-NEXT: [[LSR_IV1:%.*]] = phi i64 [ [[LSR_IV_NEXT2:%.*]], [[LOOP]] ], [ -1, [[ENTRY:%.*]] ]
-; CHECK-NEXT: [[LSR_IV:%.*]] = phi i64 [ [[LSR_IV_NEXT:%.*]], [[LOOP]] ], [ 2, [[ENTRY]] ]
+; CHECK-NEXT: [[LSR_IV:%.*]] = phi i64 [ [[LSR_IV_NEXT:%.*]], [[LOOP]] ], [ 2, [[ENTRY:%.*]] ]
; CHECK-NEXT: [[IV_1:%.*]] = phi i32 [ 1, [[ENTRY]] ], [ [[DEC_1:%.*]], [[LOOP]] ]
-; CHECK-NEXT: [[TMP0:%.*]] = add i64 [[LSR_IV1]], 1
+; CHECK-NEXT: [[TMP0:%.*]] = phi i64 [ 0, [[ENTRY]] ], [ [[INC_1:%.*]], [[LOOP]] ]
+; CHECK-NEXT: [[INC_1]] = add nsw i64 [[TMP0]], 1
; CHECK-NEXT: [[DEC_1]] = add nsw i32 [[IV_1]], -1
; CHECK-NEXT: call void @use(i64 [[TMP0]])
; CHECK-NEXT: [[LSR_IV_NEXT]] = add nsw i64 [[LSR_IV]], -1
; CHECK-NEXT: [[TMP:%.*]] = trunc i64 [[LSR_IV_NEXT]] to i32
-; CHECK-NEXT: [[LSR_IV_NEXT2]] = add nsw i64 [[LSR_IV1]], 1
; CHECK-NEXT: [[CMP6_1:%.*]] = icmp sgt i32 [[TMP]], 0
; CHECK-NEXT: br i1 [[CMP6_1]], label [[LOOP]], label [[EXIT:%.*]]
; CHECK: exit:
; CHECK-NEXT: call void @use(i64 [[LSR_IV_NEXT]])
+; CHECK-NEXT: [[LSR_IV_NEXT2:%.*]] = add i64 [[INC_1]], -1
; CHECK-NEXT: call void @use(i64 [[LSR_IV_NEXT2]])
; CHECK-NEXT: [[TMP3:%.*]] = urem i32 [[DEC_1]], 53
; CHECK-NEXT: ret i32 [[TMP3]]
diff --git a/llvm/test/Transforms/LoopStrengthReduce/X86/pr62660-normalization-failure.ll b/llvm/test/Transforms/LoopStrengthReduce/X86/pr62660-normalization-failure.ll
index 2d9478e476cb1..e6ee9b467c5e0 100644
--- a/llvm/test/Transforms/LoopStrengthReduce/X86/pr62660-normalization-failure.ll
+++ b/llvm/test/Transforms/LoopStrengthReduce/X86/pr62660-normalization-failure.ll
@@ -9,18 +9,17 @@ define i64 @test_pr62660() {
; CHECK-NEXT: entry:
; CHECK-NEXT: br label [[LOOP:%.*]]
; CHECK: loop:
-; CHECK-NEXT: [[LSR_IV:%.*]] = phi i64 [ [[LSR_IV_NEXT:%.*]], [[LOOP]] ], [ -1, [[ENTRY:%.*]] ]
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[IV_NEXT:%.*]], [[LOOP]] ]
-; CHECK-NEXT: [[TMP0:%.*]] = add i64 [[LSR_IV]], 1
+; CHECK-NEXT: [[TMP0:%.*]] = phi i64 [ [[LSR_IV_NEXT1:%.*]], [[LOOP]] ], [ 0, [[ENTRY:%.*]] ]
; CHECK-NEXT: [[TMP:%.*]] = trunc i64 [[TMP0]] to i32
-; CHECK-NEXT: [[CONV1:%.*]] = and i32 [[TMP]], 65535
-; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[IV]], -1
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i64 [[TMP0]] to i32
+; CHECK-NEXT: [[CONV1:%.*]] = and i32 [[TMP1]], 65535
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[TMP]], -1
; CHECK-NEXT: [[SUB:%.*]] = add i32 [[ADD]], [[CONV1]]
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i32 [[IV]], 1
-; CHECK-NEXT: [[LSR_IV_NEXT]] = add nsw i64 [[LSR_IV]], 1
+; CHECK-NEXT: [[LSR_IV_NEXT1]] = add nuw nsw i64 [[TMP0]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp sgt i32 [[SUB]], 8
; CHECK-NEXT: br i1 [[CMP]], label [[LOOP]], label [[EXIT:%.*]]
; CHECK: exit:
+; CHECK-NEXT: [[LSR_IV_NEXT:%.*]] = add i64 [[LSR_IV_NEXT1]], -1
; CHECK-NEXT: ret i64 [[LSR_IV_NEXT]]
;
entry:
diff --git a/llvm/test/Transforms/LoopStrengthReduce/duplicated-phis.ll b/llvm/test/Transforms/LoopStrengthReduce/duplicated-phis.ll
index 43389b5df8f00..9e0cfb1e39f61 100644
--- a/llvm/test/Transforms/LoopStrengthReduce/duplicated-phis.ll
+++ b/llvm/test/Transforms/LoopStrengthReduce/duplicated-phis.ll
@@ -19,7 +19,7 @@ define i64 @test_duplicated_phis(i64 noundef %N) {
; CHECK-NEXT: [[UNROLL_ITER:%.*]] = and i64 [[MUL]], -4
; CHECK-NEXT: [[TMP4:%.*]] = add i64 [[UNROLL_ITER]], -4
; CHECK-NEXT: [[TMP3:%.*]] = lshr i64 [[TMP4]], 1
-; CHECK-NEXT: [[LSR_IV_NEXT:%.*]] = sub i64 -3, [[TMP3]]
+; CHECK-NEXT: [[TMP5:%.*]] = sub i64 -2, [[TMP3]]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
; CHECK-NEXT: [[I_07:%.*]] = phi i64 [ 0, %[[FOR_BODY_PREHEADER_NEW]] ], [ [[INC_3:%.*]], %[[FOR_BODY]] ]
@@ -27,11 +27,11 @@ define i64 @test_duplicated_phis(i64 noundef %N) {
; CHECK-NEXT: [[NITER_NCMP_3_NOT:%.*]] = icmp eq i64 [[UNROLL_ITER]], [[INC_3]]
; CHECK-NEXT: br i1 [[NITER_NCMP_3_NOT]], label %[[FOR_END_LOOPEXIT_UNR_LCSSA_LOOPEXIT:.*]], label %[[FOR_BODY]]
; CHECK: [[FOR_END_LOOPEXIT_UNR_LCSSA_LOOPEXIT]]:
-; CHECK-NEXT: [[TMP1:%.*]] = add i64 [[LSR_IV_NEXT]], 1
+; CHECK-NEXT: [[TMP6:%.*]] = add i64 [[TMP5]], -1
; CHECK-NEXT: br label %[[FOR_END_LOOPEXIT_UNR_LCSSA]]
; CHECK: [[FOR_END_LOOPEXIT_UNR_LCSSA]]:
-; CHECK-NEXT: [[RES_1_LCSSA_PH:%.*]] = phi i64 [ undef, %[[FOR_BODY_PREHEADER]] ], [ [[TMP1]], %[[FOR_END_LOOPEXIT_UNR_LCSSA_LOOPEXIT]] ]
-; CHECK-NEXT: [[RES_09_UNR:%.*]] = phi i64 [ -1, %[[FOR_BODY_PREHEADER]] ], [ [[LSR_IV_NEXT]], %[[FOR_END_LOOPEXIT_UNR_LCSSA_LOOPEXIT]] ]
+; CHECK-NEXT: [[RES_1_LCSSA_PH:%.*]] = phi i64 [ undef, %[[FOR_BODY_PREHEADER]] ], [ [[TMP5]], %[[FOR_END_LOOPEXIT_UNR_LCSSA_LOOPEXIT]] ]
+; CHECK-NEXT: [[RES_09_UNR:%.*]] = phi i64 [ -1, %[[FOR_BODY_PREHEADER]] ], [ [[TMP6]], %[[FOR_END_LOOPEXIT_UNR_LCSSA_LOOPEXIT]] ]
; CHECK-NEXT: [[TMP2:%.*]] = and i64 [[N]], 1
; CHECK-NEXT: [[LCMP_MOD_NOT:%.*]] = icmp eq i64 [[TMP2]], 0
; CHECK-NEXT: [[SPEC_SELECT:%.*]] = select i1 [[LCMP_MOD_NOT]], i64 [[RES_1_LCSSA_PH]], i64 [[RES_09_UNR]]
More information about the llvm-commits
mailing list