[llvm] [LV] Bail out if loop nest contains non-widenable types. (PR #226463)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Thu Oct 1 06:19:42 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/226463
>From f97003d49877f7866cccc087a466854a56c95c71 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Thu, 24 Sep 2026 19:01:55 +0100
Subject: [PATCH 1/3] [LV] Bail out if loop nest contains non-widenable types.
Currently all instructions are widened during outer-loop vectorization.
That requires the result types of all instructions to be widen-able.
Update canVectorizeOuterLoop to check the types of all instructions, via
a shared helper.
---
.../Vectorize/LoopVectorizationLegality.cpp | 52 ++-
.../outer-loop-unvectorizable-types.ll | 313 ++++++++++++++++++
2 files changed, 347 insertions(+), 18 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/outer-loop-unvectorizable-types.ll
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index 862470ce0de6d..5813738e78024 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -615,6 +615,22 @@ bool LoopVectorizationLegality::isUniformMemOp(
return isUniform(Ptr, VF) && !blockNeedsPredication(I.getParent());
}
+/// Returns true if the type produced by \p I can be widened. Casts from vector
+/// types and extractelement instructions cannot be widened. Struct results are
+/// only supported if \p AllowStructCalls is set, for calls whose users are all
+/// extractvalue instructions and whose struct element types can be widened.
+static bool canWidenResultType(const Instruction &I, bool AllowStructCalls) {
+ if (isa<ExtractElementInst>(I) ||
+ (isa<CastInst>(I) &&
+ !VectorType::isValidElementType(I.getOperand(0)->getType())))
+ return false;
+ Type *Ty = I.getType();
+ if (!isa<StructType>(Ty))
+ return canVectorizeTy(Ty);
+ return AllowStructCalls && isa<CallInst>(I) && canVectorizeTy(Ty) &&
+ all_of(I.users(), IsaPred<ExtractValueInst>);
+}
+
bool LoopVectorizationLegality::canVectorizeOuterLoop() {
assert(!TheLoop->isInnermost() && "We are not vectorizing an outer loop.");
// Store the result and return it at the end instead of exiting early, in case
@@ -623,6 +639,23 @@ bool LoopVectorizationLegality::canVectorizeOuterLoop() {
bool DoExtraAnalysis = ORE->allowExtraAnalysis(DEBUG_TYPE);
for (BasicBlock *BB : TheLoop->blocks()) {
+ // Instructions in the loop nest are widened, so the types they produce and
+ // store must be widenable. Struct-returning calls are not supported yet.
+ for (Instruction &I : *BB) {
+ auto *SI = dyn_cast<StoreInst>(&I);
+ if (canWidenResultType(I, /*AllowStructCalls=*/false) &&
+ (!SI ||
+ VectorType::isValidElementType(SI->getValueOperand()->getType())))
+ continue;
+ reportVectorizationFailure("Found unvectorizable type",
+ "instruction type cannot be vectorized",
+ "CantVectorizeInstructionType", ORE, TheLoop,
+ &I);
+ if (!DoExtraAnalysis)
+ return false;
+ Result = false;
+ }
+
// Check whether the BB terminator is a branch. Any other terminator is
// not supported yet.
Instruction *Term = BB->getTerminator();
@@ -974,25 +1007,8 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
if (CI && !VFDatabase::getMappings(*CI).empty())
VecCallVariantsFound = true;
- auto CanWidenInstructionTy = [](Instruction const &Inst) {
- Type *InstTy = Inst.getType();
- if (!isa<StructType>(InstTy))
- return canVectorizeTy(InstTy);
-
- // For now, we only recognize struct values returned from calls where
- // all users are extractvalue as vectorizable. All element types of the
- // struct must be types that can be widened.
- return isa<CallInst>(Inst) && canVectorizeTy(InstTy) &&
- all_of(Inst.users(), IsaPred<ExtractValueInst>);
- };
-
// Check that the instruction return type is vectorizable.
- // We can't vectorize casts from vector type to scalar type.
- // Also, we can't vectorize extractelement instructions.
- if (!CanWidenInstructionTy(I) ||
- (isa<CastInst>(I) &&
- !VectorType::isValidElementType(I.getOperand(0)->getType())) ||
- isa<ExtractElementInst>(I)) {
+ if (!canWidenResultType(I, /*AllowStructCalls=*/true)) {
reportVectorizationFailure("Found unvectorizable type",
"instruction return type cannot be vectorized",
"CantVectorizeInstructionReturnType", ORE,
diff --git a/llvm/test/Transforms/LoopVectorize/outer-loop-unvectorizable-types.ll b/llvm/test/Transforms/LoopVectorize/outer-loop-unvectorizable-types.ll
new file mode 100644
index 0000000000000..33e2379a64e02
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/outer-loop-unvectorizable-types.ll
@@ -0,0 +1,313 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -passes=loop-vectorize -enable-vplan-native-path -force-vector-width=4 -S %s | FileCheck %s
+
+; Outer loops with values whose types cannot be widened must not be vectorized.
+
+define void @vector_typed_load_phi_store(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @vector_typed_load_phi_store(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[OUTER_HEADER:.*]]
+; CHECK: [[OUTER_HEADER]]:
+; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT: br label %[[INNER:.*]]
+; CHECK: [[INNER]]:
+; CHECK-NEXT: [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT: [[RED:%.*]] = phi <2 x i32> [ zeroinitializer, %[[OUTER_HEADER]] ], [ [[RED_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds <2 x i32>, ptr [[B]], i64 [[INNER_IV]]
+; CHECK-NEXT: [[L_B:%.*]] = load <2 x i32>, ptr [[GEP_B]], align 8
+; CHECK-NEXT: [[RED_NEXT]] = add <2 x i32> [[RED]], [[L_B]]
+; CHECK-NEXT: [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
+; CHECK-NEXT: [[INNER_EC:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], [[M]]
+; CHECK-NEXT: br i1 [[INNER_EC]], label %[[OUTER_LATCH]], label %[[INNER]]
+; CHECK: [[OUTER_LATCH]]:
+; CHECK-NEXT: [[RED_LCSSA:%.*]] = phi <2 x i32> [ [[RED_NEXT]], %[[INNER]] ]
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds <2 x i32>, ptr [[A]], i64 [[OUTER_IV]]
+; CHECK-NEXT: store <2 x i32> [[RED_LCSSA]], ptr [[GEP_A]], align 8
+; CHECK-NEXT: [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
+; CHECK-NEXT: [[OUTER_EC:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[OUTER_EC]], label %[[EXIT:.*]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %outer.header
+
+outer.header:
+ %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+ br label %inner
+
+inner:
+ %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner ]
+ %red = phi <2 x i32> [ zeroinitializer, %outer.header ], [ %red.next, %inner ]
+ %gep.B = getelementptr inbounds <2 x i32>, ptr %B, i64 %inner.iv
+ %l.B = load <2 x i32>, ptr %gep.B, align 8
+ %red.next = add <2 x i32> %red, %l.B
+ %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+ %inner.ec = icmp eq i64 %inner.iv.next, %M
+ br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+ %red.lcssa = phi <2 x i32> [ %red.next, %inner ]
+ %gep.A = getelementptr inbounds <2 x i32>, ptr %A, i64 %outer.iv
+ store <2 x i32> %red.lcssa, ptr %gep.A, align 8
+ %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+ %outer.ec = icmp eq i64 %outer.iv.next, %N
+ br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+ ret void
+}
+
+define void @store_vector_live_in(ptr noalias %A, <2 x i32> %v, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @store_vector_live_in(
+; CHECK-SAME: ptr noalias [[A:%.*]], <2 x i32> [[V:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[OUTER_HEADER:.*]]
+; CHECK: [[OUTER_HEADER]]:
+; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT: br label %[[INNER:.*]]
+; CHECK: [[INNER]]:
+; CHECK-NEXT: [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT: [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
+; CHECK-NEXT: [[INNER_EC:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], [[M]]
+; CHECK-NEXT: br i1 [[INNER_EC]], label %[[OUTER_LATCH]], label %[[INNER]]
+; CHECK: [[OUTER_LATCH]]:
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds <2 x i32>, ptr [[A]], i64 [[OUTER_IV]]
+; CHECK-NEXT: store <2 x i32> [[V]], ptr [[GEP_A]], align 8
+; CHECK-NEXT: [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
+; CHECK-NEXT: [[OUTER_EC:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[OUTER_EC]], label %[[EXIT:.*]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP0]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %outer.header
+
+outer.header:
+ %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+ br label %inner
+
+inner:
+ %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner ]
+ %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+ %inner.ec = icmp eq i64 %inner.iv.next, %M
+ br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+ %gep.A = getelementptr inbounds <2 x i32>, ptr %A, i64 %outer.iv
+ store <2 x i32> %v, ptr %gep.A, align 8
+ %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+ %outer.ec = icmp eq i64 %outer.iv.next, %N
+ br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+ ret void
+}
+
+define void @extractelement_live_in(ptr noalias %A, <2 x i32> %v, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @extractelement_live_in(
+; CHECK-SAME: ptr noalias [[A:%.*]], <2 x i32> [[V:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[OUTER_HEADER:.*]]
+; CHECK: [[OUTER_HEADER]]:
+; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT: br label %[[INNER:.*]]
+; CHECK: [[INNER]]:
+; CHECK-NEXT: [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT: [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
+; CHECK-NEXT: [[INNER_EC:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], [[M]]
+; CHECK-NEXT: br i1 [[INNER_EC]], label %[[OUTER_LATCH]], label %[[INNER]]
+; CHECK: [[OUTER_LATCH]]:
+; CHECK-NEXT: [[IDX:%.*]] = and i64 [[OUTER_IV]], 1
+; CHECK-NEXT: [[EXT:%.*]] = extractelement <2 x i32> [[V]], i64 [[IDX]]
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[OUTER_IV]]
+; CHECK-NEXT: store i32 [[EXT]], ptr [[GEP_A]], align 4
+; CHECK-NEXT: [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
+; CHECK-NEXT: [[OUTER_EC:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[OUTER_EC]], label %[[EXIT:.*]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP0]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %outer.header
+
+outer.header:
+ %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+ br label %inner
+
+inner:
+ %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner ]
+ %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+ %inner.ec = icmp eq i64 %inner.iv.next, %M
+ br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+ %idx = and i64 %outer.iv, 1
+ %ext = extractelement <2 x i32> %v, i64 %idx
+ %gep.A = getelementptr inbounds i32, ptr %A, i64 %outer.iv
+ store i32 %ext, ptr %gep.A, align 4
+ %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+ %outer.ec = icmp eq i64 %outer.iv.next, %N
+ br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+ ret void
+}
+
+define void @cast_from_vector_live_in(ptr noalias %A, <2 x i32> %v, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @cast_from_vector_live_in(
+; CHECK-SAME: ptr noalias [[A:%.*]], <2 x i32> [[V:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[OUTER_HEADER:.*]]
+; CHECK: [[OUTER_HEADER]]:
+; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT: [[BC:%.*]] = bitcast <2 x i32> [[V]] to i64
+; CHECK-NEXT: [[ADD:%.*]] = add i64 [[BC]], [[OUTER_IV]]
+; CHECK-NEXT: br label %[[INNER:.*]]
+; CHECK: [[INNER]]:
+; CHECK-NEXT: [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT: [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
+; CHECK-NEXT: [[INNER_EC:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], [[M]]
+; CHECK-NEXT: br i1 [[INNER_EC]], label %[[OUTER_LATCH]], label %[[INNER]]
+; CHECK: [[OUTER_LATCH]]:
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[OUTER_IV]]
+; CHECK-NEXT: store i64 [[ADD]], ptr [[GEP_A]], align 8
+; CHECK-NEXT: [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
+; CHECK-NEXT: [[OUTER_EC:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[OUTER_EC]], label %[[EXIT:.*]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP0]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %outer.header
+
+outer.header:
+ %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+ %bc = bitcast <2 x i32> %v to i64
+ %add = add i64 %bc, %outer.iv
+ br label %inner
+
+inner:
+ %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner ]
+ %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+ %inner.ec = icmp eq i64 %inner.iv.next, %M
+ br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+ %gep.A = getelementptr inbounds i64, ptr %A, i64 %outer.iv
+ store i64 %add, ptr %gep.A, align 8
+ %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+ %outer.ec = icmp eq i64 %outer.iv.next, %N
+ br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+ ret void
+}
+
+define void @struct_typed_load(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @struct_typed_load(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[OUTER_HEADER:.*]]
+; CHECK: [[OUTER_HEADER]]:
+; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds { i32, i32 }, ptr [[B]], i64 [[OUTER_IV]]
+; CHECK-NEXT: [[L_B:%.*]] = load { i32, i32 }, ptr [[GEP_B]], align 4
+; CHECK-NEXT: [[EXT:%.*]] = extractvalue { i32, i32 } [[L_B]], 1
+; CHECK-NEXT: br label %[[INNER:.*]]
+; CHECK: [[INNER]]:
+; CHECK-NEXT: [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT: [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
+; CHECK-NEXT: [[INNER_EC:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], [[M]]
+; CHECK-NEXT: br i1 [[INNER_EC]], label %[[OUTER_LATCH]], label %[[INNER]]
+; CHECK: [[OUTER_LATCH]]:
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[OUTER_IV]]
+; CHECK-NEXT: store i32 [[EXT]], ptr [[GEP_A]], align 4
+; CHECK-NEXT: [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
+; CHECK-NEXT: [[OUTER_EC:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[OUTER_EC]], label %[[EXIT:.*]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP0]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %outer.header
+
+outer.header:
+ %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+ %gep.B = getelementptr inbounds { i32, i32 }, ptr %B, i64 %outer.iv
+ %l.B = load { i32, i32 }, ptr %gep.B, align 4
+ %ext = extractvalue { i32, i32 } %l.B, 1
+ br label %inner
+
+inner:
+ %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner ]
+ %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+ %inner.ec = icmp eq i64 %inner.iv.next, %M
+ br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+ %gep.A = getelementptr inbounds i32, ptr %A, i64 %outer.iv
+ store i32 %ext, ptr %gep.A, align 4
+ %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+ %outer.ec = icmp eq i64 %outer.iv.next, %N
+ br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+ ret void
+}
+
+define void @struct_returning_intrinsic(ptr noalias %A, i32 %x, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @struct_returning_intrinsic(
+; CHECK-SAME: ptr noalias [[A:%.*]], i32 [[X:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[OUTER_HEADER:.*]]
+; CHECK: [[OUTER_HEADER]]:
+; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT: br label %[[INNER:.*]]
+; CHECK: [[INNER]]:
+; CHECK-NEXT: [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT: [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
+; CHECK-NEXT: [[INNER_EC:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], [[M]]
+; CHECK-NEXT: br i1 [[INNER_EC]], label %[[OUTER_LATCH]], label %[[INNER]]
+; CHECK: [[OUTER_LATCH]]:
+; CHECK-NEXT: [[TRUNC:%.*]] = trunc i64 [[OUTER_IV]] to i32
+; CHECK-NEXT: [[SADD:%.*]] = call { i32, i1 } @llvm.sadd.with.overflow.i32(i32 [[TRUNC]], i32 [[X]])
+; CHECK-NEXT: [[EXT:%.*]] = extractvalue { i32, i1 } [[SADD]], 0
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[OUTER_IV]]
+; CHECK-NEXT: store i32 [[EXT]], ptr [[GEP_A]], align 4
+; CHECK-NEXT: [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
+; CHECK-NEXT: [[OUTER_EC:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[OUTER_EC]], label %[[EXIT:.*]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP0]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %outer.header
+
+outer.header:
+ %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+ br label %inner
+
+inner:
+ %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner ]
+ %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+ %inner.ec = icmp eq i64 %inner.iv.next, %M
+ br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+ %trunc = trunc i64 %outer.iv to i32
+ %sadd = call { i32, i1 } @llvm.sadd.with.overflow.i32(i32 %trunc, i32 %x)
+ %ext = extractvalue { i32, i1 } %sadd, 0
+ %gep.A = getelementptr inbounds i32, ptr %A, i64 %outer.iv
+ store i32 %ext, ptr %gep.A, align 4
+ %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+ %outer.ec = icmp eq i64 %outer.iv.next, %N
+ br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+ ret void
+}
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.vectorize.enable"}
>From 25dbe6a014e130150714692c897f76ee4ea081a5 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Wed, 30 Sep 2026 11:46:50 +0100
Subject: [PATCH 2/3] !fixup move reporting logic to canWidenTypes
---
.../Vectorize/LoopVectorizationLegality.cpp | 49 ++++++++++---------
1 file changed, 25 insertions(+), 24 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index ce2d4dceb3226..9b31d74e76f66 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -631,6 +631,26 @@ static bool canWidenResultType(const Instruction &I, bool AllowStructCalls) {
all_of(I.users(), IsaPred<ExtractValueInst>);
}
+/// Returns true if the types produced and stored by \p I can be widened,
+/// otherwise reports a vectorization failure for \p TheLoop and returns false.
+static bool canWidenTypes(Instruction &I, bool AllowStructCalls,
+ OptimizationRemarkEmitter *ORE, Loop *TheLoop) {
+ if (!canWidenResultType(I, AllowStructCalls)) {
+ reportVectorizationFailure("Found unvectorizable type",
+ "instruction return type cannot be vectorized",
+ "CantVectorizeInstructionReturnType", ORE,
+ TheLoop, &I);
+ return false;
+ }
+ auto *SI = dyn_cast<StoreInst>(&I);
+ if (SI && !VectorType::isValidElementType(SI->getValueOperand()->getType())) {
+ reportVectorizationFailure("Store instruction cannot be vectorized",
+ "CantVectorizeStore", ORE, TheLoop, SI);
+ return false;
+ }
+ return true;
+}
+
bool LoopVectorizationLegality::canVectorizeOuterLoop() {
assert(!TheLoop->isInnermost() && "We are not vectorizing an outer loop.");
// Store the result and return it at the end instead of exiting early, in case
@@ -642,15 +662,8 @@ bool LoopVectorizationLegality::canVectorizeOuterLoop() {
// Instructions in the loop nest are widened, so the types they produce and
// store must be widenable. Struct-returning calls are not supported yet.
for (Instruction &I : *BB) {
- auto *SI = dyn_cast<StoreInst>(&I);
- if (canWidenResultType(I, /*AllowStructCalls=*/false) &&
- (!SI ||
- VectorType::isValidElementType(SI->getValueOperand()->getType())))
+ if (canWidenTypes(I, /*AllowStructCalls=*/false, ORE, TheLoop))
continue;
- reportVectorizationFailure("Found unvectorizable type",
- "instruction type cannot be vectorized",
- "CantVectorizeInstructionType", ORE, TheLoop,
- &I);
if (!DoExtraAnalysis)
return false;
Result = false;
@@ -1007,29 +1020,17 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
if (CI && !VFDatabase::getMappings(*CI).empty())
VecCallVariantsFound = true;
- // Check that the instruction return type is vectorizable.
- if (!canWidenResultType(I, /*AllowStructCalls=*/true)) {
- reportVectorizationFailure("Found unvectorizable type",
- "instruction return type cannot be vectorized",
- "CantVectorizeInstructionReturnType", ORE,
- TheLoop, &I);
+ // Check that the instruction return and stored types are vectorizable.
+ if (!canWidenTypes(I, /*AllowStructCalls=*/true, ORE, TheLoop))
return false;
- }
- // Check that the stored type is vectorizable.
if (auto *ST = dyn_cast<StoreInst>(&I)) {
- Type *T = ST->getValueOperand()->getType();
- if (!VectorType::isValidElementType(T)) {
- reportVectorizationFailure("Store instruction cannot be vectorized",
- "CantVectorizeStore", ORE, TheLoop, ST);
- return false;
- }
-
// For nontemporal stores, check that a nontemporal vector version is
// supported on the target.
if (ST->getMetadata(LLVMContext::MD_nontemporal)) {
// Arbitrarily try a vector of 2 elements.
- auto *VecTy = FixedVectorType::get(T, /*NumElts=*/2);
+ auto *VecTy =
+ FixedVectorType::get(ST->getValueOperand()->getType(), /*NumElts=*/2);
assert(VecTy && "did not find vectorized version of stored type");
if (!TTI->isLegalNTStore(VecTy, ST->getAlign())) {
reportVectorizationFailure(
>From 9743ef554d8f0090630a12eb2b99128a79a384ca Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Wed, 30 Sep 2026 11:57:18 +0100
Subject: [PATCH 3/3] !fixup add remark test
---
.../outer-loop-legality-remarks.ll | 82 +++++++++++++++++++
1 file changed, 82 insertions(+)
create mode 100644 llvm/test/Transforms/LoopVectorize/outer-loop-legality-remarks.ll
diff --git a/llvm/test/Transforms/LoopVectorize/outer-loop-legality-remarks.ll b/llvm/test/Transforms/LoopVectorize/outer-loop-legality-remarks.ll
new file mode 100644
index 0000000000000..00b50da2750d8
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/outer-loop-legality-remarks.ll
@@ -0,0 +1,82 @@
+; RUN: opt -passes=loop-vectorize -enable-vplan-native-path -disable-output \
+; RUN: -pass-remarks-output=- %s | FileCheck %s --match-full-lines \
+; RUN: --implicit-check-not='--- !'
+
+; CHECK: --- !Analysis
+; CHECK-NEXT: Pass: loop-vectorize
+; CHECK-NEXT: Name: CantVectorizeInstructionReturnType
+; CHECK-NEXT: DebugLoc: { File: test.c, Line: 3, Column: 0 }
+; CHECK-NEXT: Function: vector_load_store
+; CHECK-NEXT: Args:
+; CHECK-NEXT: - String: 'loop not vectorized: '
+; CHECK-NEXT: - String: instruction return type cannot be vectorized
+; CHECK-NEXT: ...
+; CHECK-NEXT: --- !Analysis
+; CHECK-NEXT: Pass: loop-vectorize
+; CHECK-NEXT: Name: CantVectorizeStore
+; CHECK-NEXT: DebugLoc: { File: test.c, Line: 4, Column: 0 }
+; CHECK-NEXT: Function: vector_load_store
+; CHECK-NEXT: Args:
+; CHECK-NEXT: - String: 'loop not vectorized: '
+; CHECK-NEXT: - String: Store instruction cannot be vectorized
+; CHECK-NEXT: ...
+; CHECK-NEXT: --- !Analysis
+; CHECK-NEXT: Pass: loop-vectorize
+; CHECK-NEXT: Name: UnsupportedOuterLoop
+; CHECK-NEXT: DebugLoc: { File: test.c, Line: 2, Column: 0 }
+; CHECK-NEXT: Function: vector_load_store
+; CHECK-NEXT: Args:
+; CHECK-NEXT: - String: 'loop not vectorized: '
+; CHECK-NEXT: - String: Unsupported outer loop
+; CHECK-NEXT: ...
+; CHECK-NEXT: --- !Missed
+; CHECK-NEXT: Pass: loop-vectorize
+; CHECK-NEXT: Name: MissedDetails
+; CHECK-NEXT: DebugLoc: { File: test.c, Line: 2, Column: 0 }
+; CHECK-NEXT: Function: vector_load_store
+; CHECK-NEXT: Args:
+; CHECK-NEXT: - String: loop not vectorized
+; CHECK-NEXT: - String: ' (Force='
+; CHECK-NEXT: - Force: 'true'
+; CHECK-NEXT: - String: ')'
+; CHECK-NEXT: ...
+define void @vector_load_store(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) !dbg !3 {
+entry:
+ br label %outer.header
+
+outer.header:
+ %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+ br label %inner
+
+inner:
+ %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner ]
+ %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+ %inner.ec = icmp eq i64 %inner.iv.next, %M
+ br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+ %gep.B = getelementptr inbounds <2 x i32>, ptr %B, i64 %outer.iv
+ %l = load <2 x i32>, ptr %gep.B, align 8, !dbg !7
+ %gep.A = getelementptr inbounds <2 x i32>, ptr %A, i64 %outer.iv
+ store <2 x i32> %l, ptr %gep.A, align 8, !dbg !8
+ %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+ %outer.ec = icmp eq i64 %outer.iv.next, %N
+ br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !4
+
+exit:
+ ret void
+}
+
+!llvm.dbg.cu = !{!0}
+!llvm.module.flags = !{!2}
+
+!0 = distinct !DICompileUnit(language: DW_LANG_C11, file: !1, emissionKind: LineTablesOnly)
+!1 = !DIFile(filename: "test.c", directory: "/")
+!2 = !{i32 2, !"Debug Info Version", i32 3}
+!3 = distinct !DISubprogram(name: "vector_load_store", scope: !1, file: !1, line: 1, type: !9, spFlags: DISPFlagDefinition, unit: !0)
+!4 = distinct !{!4, !5, !6}
+!5 = !DILocation(line: 2, scope: !3)
+!6 = !{!"llvm.loop.vectorize.enable"}
+!7 = !DILocation(line: 3, scope: !3)
+!8 = !DILocation(line: 4, scope: !3)
+!9 = !DISubroutineType(types: !{})
More information about the llvm-commits
mailing list