[llvm] [VPlan] Support different loop-invariant ops in narrowIG. (PR #203785)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Sun Jun 28 12:44:00 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/203785
>From 1f33e854fd04bfcf3b2140f610d60d2596ee41b8 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sat, 13 Jun 2026 17:15:50 +0200
Subject: [PATCH 1/2] [VPlan] Support different loop-invariant ops in narrowIG.
Generalize the distinct per-field operand narrowing in
narrowInterleaveGroups from plain live-ins to any value defined outside
the vector loop region.
Depends on https://github.com/llvm/llvm-project/pull/203778 (included in
PR)
---
.../Transforms/Vectorize/VPlanTransforms.cpp | 18 ++++++------
...interleave-to-widen-memory-constant-ops.ll | 28 ++++++-------------
2 files changed, 17 insertions(+), 29 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index bfad5d02d1767..eebf3cdaea7d7 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -5760,15 +5760,15 @@ VPlanTransforms::expandSCEVs(VPlan &Plan, ScalarEvolution &SE) {
static bool canNarrowLoad(VPSingleDefRecipe *WideMember0, unsigned OpIdx,
VPValue *OpV, unsigned Idx, bool IsScalable) {
VPValue *Member0Op = WideMember0->getOperand(OpIdx);
- VPRecipeBase *Member0OpR = Member0Op->getDefiningRecipe();
- if (!Member0OpR) {
- // Member0's operand is a uniform live-in, broadcast across all fields.
+ if (Member0Op->isDefinedOutsideLoopRegions()) {
+ // Operand matches Member0, broadcast across all fields.
if (Member0Op == OpV)
return true;
- // Otherwise distinct per-field live-ins are assembled into a BuildVector.
- return !IsScalable && !OpV->hasDefiningRecipe() &&
+ // Otherwise distinct per-field VPValues are assembled into a BuildVector.
+ return !IsScalable && OpV->isDefinedOutsideLoopRegions() &&
OpV->getScalarType() == Member0Op->getScalarType();
}
+ VPRecipeBase *Member0OpR = Member0Op->getDefiningRecipe();
if (auto *W = dyn_cast<VPWidenLoadRecipe>(Member0OpR))
// For scalable VFs, the narrowed plan processes vscale iterations at once,
// so a shared wide load cannot be narrowed to a uniform scalar; bail out.
@@ -5873,17 +5873,16 @@ static VPValue *narrowInterleaveGroupOp(ArrayRef<VPValue *> Members,
SmallPtrSetImpl<VPValue *> &NarrowedOps,
VPBasicBlock *Preheader) {
VPValue *V = Members.front();
- auto *R = V->getDefiningRecipe();
if (NarrowedOps.contains(V))
return V;
- if (!R) {
+ if (V->isDefinedOutsideLoopRegions()) {
assert(all_of(Members,
[V](VPValue *M) {
- return !M->hasDefiningRecipe() &&
+ return M->isDefinedOutsideLoopRegions() &&
M->getScalarType() == V->getScalarType();
}) &&
- "expected distinct live-ins of matching scalar type");
+ "expected distinct loop-invariant values of matching scalar type");
auto *BV = new VPInstruction(VPInstruction::BuildVector, Members);
Preheader->appendRecipe(BV);
NarrowedOps.insert(BV);
@@ -5893,6 +5892,7 @@ static VPValue *narrowInterleaveGroupOp(ArrayRef<VPValue *> Members,
if (isAlreadyNarrow(V))
return V;
+ auto *R = V->getDefiningRecipe();
if (isa<VPWidenRecipe, VPWidenCastRecipe>(R)) {
auto *WideMember0 = cast<VPRecipeWithIRFlags>(R);
for (VPValue *Member : Members.drop_front())
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-constant-ops.ll b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-constant-ops.ll
index 66559c3c3697a..9945874df559c 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-constant-ops.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-constant-ops.ll
@@ -441,36 +441,24 @@ define void @test_add_double_different_invariant_recipe_args(ptr %res, ptr noali
; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[TMP0:%.*]] = fmul double [[X]], [[Z]]
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <2 x double> poison, double [[TMP0]], i64 0
-; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <2 x double> [[BROADCAST_SPLATINSERT]], <2 x double> poison, <2 x i32> zeroinitializer
; CHECK-NEXT: [[TMP1:%.*]] = fmul double [[Y]], [[Z]]
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT5:%.*]] = insertelement <2 x double> poison, double [[TMP1]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT6:%.*]] = shufflevector <2 x double> [[BROADCAST_SPLATINSERT5]], <2 x double> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <2 x double> poison, double [[TMP0]], i32 0
+; CHECK-NEXT: [[BROADCAST_SPLAT6:%.*]] = insertelement <2 x double> [[TMP2]], double [[TMP1]], i32 1
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP4:%.*]] = add i64 [[INDEX]], 2
+; CHECK-NEXT: [[TMP4:%.*]] = add i64 [[INDEX]], 1
; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds nuw { double, double }, ptr [[A]], i64 [[INDEX]]
; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds nuw { double, double }, ptr [[A]], i64 [[TMP4]]
-; CHECK-NEXT: [[WIDE_VEC:%.*]] = load <4 x double>, ptr [[TMP5]], align 4
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = shufflevector <4 x double> [[WIDE_VEC]], <4 x double> poison, <2 x i32> <i32 0, i32 2>
-; CHECK-NEXT: [[STRIDED_VEC1:%.*]] = shufflevector <4 x double> [[WIDE_VEC]], <4 x double> poison, <2 x i32> <i32 1, i32 3>
-; CHECK-NEXT: [[WIDE_VEC2:%.*]] = load <4 x double>, ptr [[TMP6]], align 4
-; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = shufflevector <4 x double> [[WIDE_VEC2]], <4 x double> poison, <2 x i32> <i32 0, i32 2>
-; CHECK-NEXT: [[STRIDED_VEC4:%.*]] = shufflevector <4 x double> [[WIDE_VEC2]], <4 x double> poison, <2 x i32> <i32 1, i32 3>
-; CHECK-NEXT: [[TMP7:%.*]] = fadd <2 x double> [[TMP3]], [[WIDE_LOAD]]
-; CHECK-NEXT: [[TMP8:%.*]] = fadd <2 x double> [[TMP3]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[STRIDED_VEC1:%.*]] = load <2 x double>, ptr [[TMP5]], align 4
+; CHECK-NEXT: [[STRIDED_VEC4:%.*]] = load <2 x double>, ptr [[TMP6]], align 4
; CHECK-NEXT: [[TMP13:%.*]] = fadd <2 x double> [[BROADCAST_SPLAT6]], [[STRIDED_VEC1]]
; CHECK-NEXT: [[TMP14:%.*]] = fadd <2 x double> [[BROADCAST_SPLAT6]], [[STRIDED_VEC4]]
; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds nuw { double, double }, ptr [[RES]], i64 [[INDEX]]
; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds nuw { double, double }, ptr [[RES]], i64 [[TMP4]]
-; CHECK-NEXT: [[TMP15:%.*]] = shufflevector <2 x double> [[TMP7]], <2 x double> [[TMP13]], <4 x i32> <i32 0, i32 1, i32 2, i32 3>
-; CHECK-NEXT: [[INTERLEAVED_VEC:%.*]] = shufflevector <4 x double> [[TMP15]], <4 x double> poison, <4 x i32> <i32 0, i32 2, i32 1, i32 3>
-; CHECK-NEXT: store <4 x double> [[INTERLEAVED_VEC]], ptr [[TMP9]], align 4
-; CHECK-NEXT: [[TMP12:%.*]] = shufflevector <2 x double> [[TMP8]], <2 x double> [[TMP14]], <4 x i32> <i32 0, i32 1, i32 2, i32 3>
-; CHECK-NEXT: [[INTERLEAVED_VEC7:%.*]] = shufflevector <4 x double> [[TMP12]], <4 x double> poison, <4 x i32> <i32 0, i32 2, i32 1, i32 3>
-; CHECK-NEXT: store <4 x double> [[INTERLEAVED_VEC7]], ptr [[TMP10]], align 4
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: store <2 x double> [[TMP13]], ptr [[TMP9]], align 4
+; CHECK-NEXT: store <2 x double> [[TMP14]], ptr [[TMP10]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 2
; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], 100
; CHECK-NEXT: br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
>From ece7bdf42814c89d7a7e6c755974f54b6884367a Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sun, 28 Jun 2026 20:42:59 +0100
Subject: [PATCH 2/2] !fixup address comments
---
llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp | 7 +++++--
1 file changed, 5 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index a8fcb6ba1319f..eceede0e0e79b 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -5750,11 +5750,14 @@ VPlanTransforms::expandSCEVs(VPlan &Plan, ScalarEvolution &SE) {
/// must be the operand at index \p OpIdx for both the recipe at lane 0, \p
/// WideMember0). A VPInterleaveRecipe can be narrowed to a wide load, if \p V
/// is defined at \p Idx of a load interleave group.
+/// A live-in or recipe defined outside the loop region can be converted, if it
+/// is the same across all lanes, or we can create a BuildVector for it.
static bool canNarrowLoad(VPSingleDefRecipe *WideMember0, unsigned OpIdx,
VPValue *OpV, unsigned Idx, bool IsScalable) {
VPValue *Member0Op = WideMember0->getOperand(OpIdx);
if (Member0Op->isDefinedOutsideLoopRegions()) {
- // Operand matches Member0, broadcast across all fields.
+ // Operand matches Member0, broadcast across all fields for both live-ins
+ // and recipes.
if (Member0Op == OpV)
return true;
// Otherwise distinct per-field VPValues are assembled into a BuildVector.
@@ -5889,7 +5892,7 @@ static VPValue *narrowInterleaveGroupOp(ArrayRef<VPValue *> Members,
if (isAlreadyNarrow(V))
return V;
- auto *R = V->getDefiningRecipe();
+ VPRecipeBase *R = V->getDefiningRecipe();
if (isa<VPWidenRecipe, VPWidenCastRecipe>(R)) {
auto *WideMember0 = cast<VPRecipeWithIRFlags>(R);
for (VPValue *Member : Members.drop_front())
More information about the llvm-commits
mailing list