[llvm] a498e64 - [LV] Cost EVL width-adjusting zext/trunc as free (#225016)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 01:08:31 PDT 2026
Author: Pengcheng Wang
Date: 2026-09-24T08:08:21Z
New Revision: a498e642af4aded70fd48bda68b0692a58425e86
URL: https://github.com/llvm/llvm-project/commit/a498e642af4aded70fd48bda68b0692a58425e86
DIFF: https://github.com/llvm/llvm-project/commit/a498e642af4aded70fd48bda68b0692a58425e86.diff
LOG: [LV] Cost EVL width-adjusting zext/trunc as free (#225016)
With EVL tail folding, the scalar zext/trunc that adjusts the i32
ExplicitVectorLength to the canonical IV type was priced as a regular
cast (cost 1). It never lowers to an instruction, only feeding the
(free) IV increment and AVL decrement, so the phantom cost dropped
borderline loops such as TSVC s351 on RV64 to VF 1.
Here we return 0 for such casts so that we don't over-estimate the
cost.
Fixes #224987
Assisted-by: TRAE CLI (Opus 4.8)
Added:
Modified:
llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
llvm/test/Transforms/LoopVectorize/RISCV/force-vect-msg.ll
llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-cost.ll
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
index a4bfab11055e6..bacffdef18fbd 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
@@ -593,8 +593,12 @@ m_ZExtOrSelf(const Op0_t &Op0) {
return m_CombineOr(m_ZExt(Op0), Op0);
}
+template <typename Op0_t> inline auto m_ZExtOrTrunc(const Op0_t &Op0) {
+ return m_CombineOr(m_ZExt(Op0), m_Trunc(Op0));
+}
+
template <typename Op0_t> inline auto m_ZExtOrTruncOrSelf(const Op0_t &Op0) {
- return m_CombineOr(m_ZExt(Op0), m_Trunc(Op0), Op0);
+ return m_CombineOr(m_ZExtOrTrunc(Op0), Op0);
}
template <unsigned Opcode, typename Op0_t, typename Op1_t>
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 38c94512f0546..a79e8220a5b3d 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1360,9 +1360,15 @@ InstructionCost VPInstruction::computeCost(ElementCount VF,
// NOTE: At the moment it seems only possible to expose this path for
// the trunc, zext and sext opcodes.
// TODO: Update VF arg to use onlyFirstLaneUsed once WidenCast is unified.
- if (Instruction::isCast(getOpcode()))
+ if (Instruction::isCast(getOpcode())) {
+ // A scalar zext/trunc that only adjusts the width of an
+ // ExplicitVectorLength to the canonical IV type is free: it feeds only
+ // the IV increment and AVL decrement, which are modeled as free below.
+ if (match(this, m_ZExtOrTrunc(m_EVL(m_VPValue()))))
+ return 0;
return getCostForRecipeWithOpcode(getOpcode(), ElementCount::getFixed(1),
Ctx);
+ }
if (Instruction::isBinaryOp(getOpcode())) {
if (!getUnderlyingValue() && getOpcode() != Instruction::FMul) {
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/force-vect-msg.ll b/llvm/test/Transforms/LoopVectorize/RISCV/force-vect-msg.ll
index 22560793e4504..4292317345dd4 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/force-vect-msg.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/force-vect-msg.ll
@@ -3,8 +3,8 @@
; CHECK: LV: Loop hints: force=enabled
; CHECK: LV: Scalar loop costs: 4.
-; ChosenFactor.Cost is 11, but the real cost will be divided by the width, which is 2.8
-; CHECK: Cost for VF vscale x 2: 9
+; ChosenFactor.Cost is 7, but the real cost will be divided by the width, which is 1.75
+; CHECK: Cost for VF vscale x 2: 7
; Regardless of force vectorization or not, this loop will eventually be vectorized because of the cost model.
; Therefore, the following message does not need to be printed even if vectorization is explicitly forced in the metadata.
; CHECK-NOT: LV: Vectorization seems to be not beneficial, but was forced by a user.
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-cost.ll b/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-cost.ll
index 4bd2072a6d680..bb818dc4e988d 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/tail-folding-cost.ll
@@ -14,8 +14,11 @@
; DATA: Cost of 8 for VF vscale x 4: EMIT{{.*}} = active lane mask
; EVL: Cost of 1 for VF vscale x 1: EMIT{{.*}} = EXPLICIT-VECTOR-LENGTH
+; EVL: Cost of 0 for VF vscale x 1: EMIT-SCALAR vp<{{.*}}> = zext vp<%evl> to i64
; EVL: Cost of 1 for VF vscale x 2: EMIT{{.*}} = EXPLICIT-VECTOR-LENGTH
+; EVL: Cost of 0 for VF vscale x 2: EMIT-SCALAR vp<{{.*}}> = zext vp<%evl> to i64
; EVL: Cost of 1 for VF vscale x 4: EMIT{{.*}} = EXPLICIT-VECTOR-LENGTH
+; EVL: Cost of 0 for VF vscale x 4: EMIT-SCALAR vp<{{.*}}> = zext vp<%evl> to i64
define void @simple_memset(i32 %val, ptr %ptr, i64 %n) #0 {
entry:
More information about the llvm-commits
mailing list