[llvm] [VPlan] CSE ScalarIVSteps recipes (PR #191307)
Ramkumar Ramachandra via llvm-commits
llvm-commits at lists.llvm.org
Thu Apr 9 14:44:13 PDT 2026
https://github.com/artagnon created https://github.com/llvm/llvm-project/pull/191307
Extend getOpCodeOrIntrinsicID to return a pseudo opcode for ScalarIVSteps, so it can be CSE'd.
>From 64a1975f1cd541bf7542d23a75738bc9fd68daa1 Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Thu, 9 Apr 2026 22:42:22 +0100
Subject: [PATCH] [VPlan] CSE ScalarIVSteps recipes
Extend getOpCodeOrIntrinsicID to return a pseudo opcode for
ScalarIVSteps, so it can be CSE'd.
---
.../Transforms/Vectorize/VPlanTransforms.cpp | 15 ++++++-------
...-interleave-to-widen-memory-multi-block.ll | 3 +--
.../X86/cost-conditional-branches.ll | 21 ++++++-------------
.../X86/vplan-single-bit-ind-var-width-4.ll | 4 +---
4 files changed, 16 insertions(+), 27 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 098d9975beabb..906eeff862831 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -1217,13 +1217,14 @@ getOpcodeOrIntrinsicID(const VPSingleDefRecipe *R) {
.Case([](const VPWidenIntrinsicRecipe *I) {
return std::make_pair(true, I->getVectorIntrinsicID());
})
- .Case<VPVectorPointerRecipe, VPPredInstPHIRecipe>([](auto *I) {
- // For recipes that do not directly map to LLVM IR instructions,
- // assign opcodes after the last VPInstruction opcode (which is also
- // after the last IR Instruction opcode), based on the VPRecipeID.
- return std::make_pair(false,
- VPInstruction::OpsEnd + 1 + I->getVPRecipeID());
- })
+ .Case<VPVectorPointerRecipe, VPPredInstPHIRecipe, VPScalarIVStepsRecipe>(
+ [](auto *I) {
+ // For recipes that do not directly map to LLVM IR instructions,
+ // assign opcodes after the last VPInstruction opcode (which is also
+ // after the last IR Instruction opcode), based on the VPRecipeID.
+ return std::make_pair(false, VPInstruction::OpsEnd + 1 +
+ I->getVPRecipeID());
+ })
.Default([](auto *) { return std::nullopt; });
}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-multi-block.ll b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-multi-block.ll
index 8c34e6815f843..fc8b9d2e80562 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-multi-block.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-multi-block.ll
@@ -219,8 +219,7 @@ define void @load_store_interleave_group_block_var_cond(ptr noalias %data, ptr %
; VF2IC2-NEXT: [[TMP17:%.*]] = extractelement <2 x i1> [[TMP10]], i32 0
; VF2IC2-NEXT: br i1 [[TMP17]], label %[[PRED_STORE_IF9:.*]], label %[[PRED_STORE_CONTINUE10:.*]]
; VF2IC2: [[PRED_STORE_IF9]]:
-; VF2IC2-NEXT: [[TMP18:%.*]] = add i64 [[INDEX]], 2
-; VF2IC2-NEXT: [[TMP19:%.*]] = getelementptr inbounds i8, ptr [[MASKS]], i64 [[TMP18]]
+; VF2IC2-NEXT: [[TMP19:%.*]] = getelementptr inbounds i8, ptr [[MASKS]], i64 [[TMP0]]
; VF2IC2-NEXT: store i8 1, ptr [[TMP19]], align 1
; VF2IC2-NEXT: br label %[[PRED_STORE_CONTINUE10]]
; VF2IC2: [[PRED_STORE_CONTINUE10]]:
diff --git a/llvm/test/Transforms/LoopVectorize/X86/cost-conditional-branches.ll b/llvm/test/Transforms/LoopVectorize/X86/cost-conditional-branches.ll
index 00a6818918f05..c79fdd7cb0913 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/cost-conditional-branches.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/cost-conditional-branches.ll
@@ -415,9 +415,9 @@ define void @cost_duplicate_recipe_for_sinking(ptr %A, i64 %N) #2 {
; CHECK-NEXT: [[TMP9:%.*]] = shl nsw i64 [[TMP5]], 2
; CHECK-NEXT: [[TMP10:%.*]] = shl nsw i64 [[TMP6]], 2
; CHECK-NEXT: [[TMP11:%.*]] = getelementptr double, ptr [[A:%.*]], i64 [[TMP7]]
-; CHECK-NEXT: [[TMP12:%.*]] = getelementptr nusw double, ptr [[A]], i64 [[TMP8]]
-; CHECK-NEXT: [[TMP13:%.*]] = getelementptr nusw double, ptr [[A]], i64 [[TMP9]]
-; CHECK-NEXT: [[TMP14:%.*]] = getelementptr nusw double, ptr [[A]], i64 [[TMP10]]
+; CHECK-NEXT: [[TMP12:%.*]] = getelementptr double, ptr [[A]], i64 [[TMP8]]
+; CHECK-NEXT: [[TMP13:%.*]] = getelementptr double, ptr [[A]], i64 [[TMP9]]
+; CHECK-NEXT: [[TMP14:%.*]] = getelementptr double, ptr [[A]], i64 [[TMP10]]
; CHECK-NEXT: [[WIDE_VEC:%.*]] = load <16 x double>, ptr [[TMP11]], align 8
; CHECK-NEXT: [[STRIDED_VEC:%.*]] = shufflevector <16 x double> [[WIDE_VEC]], <16 x double> poison, <4 x i32> <i32 0, i32 4, i32 8, i32 12>
; CHECK-NEXT: [[WIDE_VEC1:%.*]] = load <16 x double>, ptr [[TMP12]], align 8
@@ -466,10 +466,7 @@ define void @cost_duplicate_recipe_for_sinking(ptr %A, i64 %N) #2 {
; CHECK-NEXT: [[TMP38:%.*]] = extractelement <4 x i1> [[TMP20]], i32 0
; CHECK-NEXT: br i1 [[TMP38]], label [[PRED_STORE_IF14:%.*]], label [[PRED_STORE_CONTINUE15:%.*]]
; CHECK: pred.store.if14:
-; CHECK-NEXT: [[TMP88:%.*]] = add i64 [[INDEX]], 4
-; CHECK-NEXT: [[TMP39:%.*]] = shl nsw i64 [[TMP88]], 2
-; CHECK-NEXT: [[TMP40:%.*]] = getelementptr double, ptr [[A]], i64 [[TMP39]]
-; CHECK-NEXT: store double 0.000000e+00, ptr [[TMP40]], align 8
+; CHECK-NEXT: store double 0.000000e+00, ptr [[TMP12]], align 8
; CHECK-NEXT: br label [[PRED_STORE_CONTINUE15]]
; CHECK: pred.store.continue15:
; CHECK-NEXT: [[TMP41:%.*]] = extractelement <4 x i1> [[TMP20]], i32 1
@@ -502,10 +499,7 @@ define void @cost_duplicate_recipe_for_sinking(ptr %A, i64 %N) #2 {
; CHECK-NEXT: [[TMP53:%.*]] = extractelement <4 x i1> [[TMP21]], i32 0
; CHECK-NEXT: br i1 [[TMP53]], label [[PRED_STORE_IF22:%.*]], label [[PRED_STORE_CONTINUE23:%.*]]
; CHECK: pred.store.if22:
-; CHECK-NEXT: [[TMP107:%.*]] = add i64 [[INDEX]], 8
-; CHECK-NEXT: [[TMP54:%.*]] = shl nsw i64 [[TMP107]], 2
-; CHECK-NEXT: [[TMP55:%.*]] = getelementptr double, ptr [[A]], i64 [[TMP54]]
-; CHECK-NEXT: store double 0.000000e+00, ptr [[TMP55]], align 8
+; CHECK-NEXT: store double 0.000000e+00, ptr [[TMP13]], align 8
; CHECK-NEXT: br label [[PRED_STORE_CONTINUE23]]
; CHECK: pred.store.continue23:
; CHECK-NEXT: [[TMP56:%.*]] = extractelement <4 x i1> [[TMP21]], i32 1
@@ -538,10 +532,7 @@ define void @cost_duplicate_recipe_for_sinking(ptr %A, i64 %N) #2 {
; CHECK-NEXT: [[TMP68:%.*]] = extractelement <4 x i1> [[TMP22]], i32 0
; CHECK-NEXT: br i1 [[TMP68]], label [[PRED_STORE_IF30:%.*]], label [[PRED_STORE_CONTINUE31:%.*]]
; CHECK: pred.store.if30:
-; CHECK-NEXT: [[TMP108:%.*]] = add i64 [[INDEX]], 12
-; CHECK-NEXT: [[TMP69:%.*]] = shl nsw i64 [[TMP108]], 2
-; CHECK-NEXT: [[TMP70:%.*]] = getelementptr double, ptr [[A]], i64 [[TMP69]]
-; CHECK-NEXT: store double 0.000000e+00, ptr [[TMP70]], align 8
+; CHECK-NEXT: store double 0.000000e+00, ptr [[TMP14]], align 8
; CHECK-NEXT: br label [[PRED_STORE_CONTINUE31]]
; CHECK: pred.store.continue31:
; CHECK-NEXT: [[TMP71:%.*]] = extractelement <4 x i1> [[TMP22]], i32 1
diff --git a/llvm/test/Transforms/LoopVectorize/X86/vplan-single-bit-ind-var-width-4.ll b/llvm/test/Transforms/LoopVectorize/X86/vplan-single-bit-ind-var-width-4.ll
index 0b50c856d2c57..f9bdb5dbe72ca 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/vplan-single-bit-ind-var-width-4.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/vplan-single-bit-ind-var-width-4.ll
@@ -13,14 +13,12 @@ define void @copy_bitcast_fusion(ptr noalias %foo, ptr noalias %bar) {
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[TMP2:%.*]] = select i1 true, i64 1, i64 0
; CHECK-NEXT: [[TMP3:%.*]] = select i1 false, i64 1, i64 0
-; CHECK-NEXT: [[TMP4:%.*]] = select i1 true, i64 1, i64 0
; CHECK-NEXT: [[TMP6:%.*]] = getelementptr { float, float }, ptr [[FOO]], i64 [[TMP2]]
; CHECK-NEXT: [[TMP7:%.*]] = getelementptr { float, float }, ptr [[FOO]], i64 [[TMP3]]
-; CHECK-NEXT: [[TMP8:%.*]] = getelementptr { float, float }, ptr [[FOO]], i64 [[TMP4]]
; CHECK-NEXT: [[TMP9:%.*]] = load float, ptr [[FOO]], align 4
; CHECK-NEXT: [[TMP10:%.*]] = load float, ptr [[TMP6]], align 4
; CHECK-NEXT: [[TMP11:%.*]] = load float, ptr [[TMP7]], align 4
-; CHECK-NEXT: [[TMP12:%.*]] = load float, ptr [[TMP8]], align 4
+; CHECK-NEXT: [[TMP12:%.*]] = load float, ptr [[TMP6]], align 4
; CHECK-NEXT: [[TMP13:%.*]] = insertelement <4 x float> poison, float [[TMP9]], i32 0
; CHECK-NEXT: [[TMP14:%.*]] = insertelement <4 x float> [[TMP13]], float [[TMP10]], i32 1
; CHECK-NEXT: [[TMP15:%.*]] = insertelement <4 x float> [[TMP14]], float [[TMP11]], i32 2
More information about the llvm-commits
mailing list