[llvm] [LV] Bail out if loop nest contains non-widenable types. (PR #226463)

Florian Hahn via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 25 05:08:16 PDT 2026


https://github.com/fhahn created https://github.com/llvm/llvm-project/pull/226463

Currently all instructions are widened during outer-loop vectorization.

That requires the result types of all instructions to be widen-able. Update canVectorizeOuterLoop to check the types of all instructions, via a shared helper.

>From f97003d49877f7866cccc087a466854a56c95c71 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Thu, 24 Sep 2026 19:01:55 +0100
Subject: [PATCH] [LV] Bail out if loop nest contains non-widenable types.

Currently all instructions are widened during outer-loop vectorization.

That requires the result types of all instructions to be widen-able.
Update canVectorizeOuterLoop to check the types of all instructions, via
a shared helper.
---
 .../Vectorize/LoopVectorizationLegality.cpp   |  52 ++-
 .../outer-loop-unvectorizable-types.ll        | 313 ++++++++++++++++++
 2 files changed, 347 insertions(+), 18 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/outer-loop-unvectorizable-types.ll

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index 862470ce0de6d..5813738e78024 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -615,6 +615,22 @@ bool LoopVectorizationLegality::isUniformMemOp(
   return isUniform(Ptr, VF) && !blockNeedsPredication(I.getParent());
 }
 
+/// Returns true if the type produced by \p I can be widened. Casts from vector
+/// types and extractelement instructions cannot be widened. Struct results are
+/// only supported if \p AllowStructCalls is set, for calls whose users are all
+/// extractvalue instructions and whose struct element types can be widened.
+static bool canWidenResultType(const Instruction &I, bool AllowStructCalls) {
+  if (isa<ExtractElementInst>(I) ||
+      (isa<CastInst>(I) &&
+       !VectorType::isValidElementType(I.getOperand(0)->getType())))
+    return false;
+  Type *Ty = I.getType();
+  if (!isa<StructType>(Ty))
+    return canVectorizeTy(Ty);
+  return AllowStructCalls && isa<CallInst>(I) && canVectorizeTy(Ty) &&
+         all_of(I.users(), IsaPred<ExtractValueInst>);
+}
+
 bool LoopVectorizationLegality::canVectorizeOuterLoop() {
   assert(!TheLoop->isInnermost() && "We are not vectorizing an outer loop.");
   // Store the result and return it at the end instead of exiting early, in case
@@ -623,6 +639,23 @@ bool LoopVectorizationLegality::canVectorizeOuterLoop() {
   bool DoExtraAnalysis = ORE->allowExtraAnalysis(DEBUG_TYPE);
 
   for (BasicBlock *BB : TheLoop->blocks()) {
+    // Instructions in the loop nest are widened, so the types they produce and
+    // store must be widenable. Struct-returning calls are not supported yet.
+    for (Instruction &I : *BB) {
+      auto *SI = dyn_cast<StoreInst>(&I);
+      if (canWidenResultType(I, /*AllowStructCalls=*/false) &&
+          (!SI ||
+           VectorType::isValidElementType(SI->getValueOperand()->getType())))
+        continue;
+      reportVectorizationFailure("Found unvectorizable type",
+                                 "instruction type cannot be vectorized",
+                                 "CantVectorizeInstructionType", ORE, TheLoop,
+                                 &I);
+      if (!DoExtraAnalysis)
+        return false;
+      Result = false;
+    }
+
     // Check whether the BB terminator is a branch. Any other terminator is
     // not supported yet.
     Instruction *Term = BB->getTerminator();
@@ -974,25 +1007,8 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
   if (CI && !VFDatabase::getMappings(*CI).empty())
     VecCallVariantsFound = true;
 
-  auto CanWidenInstructionTy = [](Instruction const &Inst) {
-    Type *InstTy = Inst.getType();
-    if (!isa<StructType>(InstTy))
-      return canVectorizeTy(InstTy);
-
-    // For now, we only recognize struct values returned from calls where
-    // all users are extractvalue as vectorizable. All element types of the
-    // struct must be types that can be widened.
-    return isa<CallInst>(Inst) && canVectorizeTy(InstTy) &&
-           all_of(Inst.users(), IsaPred<ExtractValueInst>);
-  };
-
   // Check that the instruction return type is vectorizable.
-  // We can't vectorize casts from vector type to scalar type.
-  // Also, we can't vectorize extractelement instructions.
-  if (!CanWidenInstructionTy(I) ||
-      (isa<CastInst>(I) &&
-       !VectorType::isValidElementType(I.getOperand(0)->getType())) ||
-      isa<ExtractElementInst>(I)) {
+  if (!canWidenResultType(I, /*AllowStructCalls=*/true)) {
     reportVectorizationFailure("Found unvectorizable type",
                                "instruction return type cannot be vectorized",
                                "CantVectorizeInstructionReturnType", ORE,
diff --git a/llvm/test/Transforms/LoopVectorize/outer-loop-unvectorizable-types.ll b/llvm/test/Transforms/LoopVectorize/outer-loop-unvectorizable-types.ll
new file mode 100644
index 0000000000000..33e2379a64e02
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/outer-loop-unvectorizable-types.ll
@@ -0,0 +1,313 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -passes=loop-vectorize -enable-vplan-native-path -force-vector-width=4 -S %s | FileCheck %s
+
+; Outer loops with values whose types cannot be widened must not be vectorized.
+
+define void @vector_typed_load_phi_store(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @vector_typed_load_phi_store(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
+; CHECK:       [[OUTER_HEADER]]:
+; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT:    br label %[[INNER:.*]]
+; CHECK:       [[INNER]]:
+; CHECK-NEXT:    [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT:    [[RED:%.*]] = phi <2 x i32> [ zeroinitializer, %[[OUTER_HEADER]] ], [ [[RED_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds <2 x i32>, ptr [[B]], i64 [[INNER_IV]]
+; CHECK-NEXT:    [[L_B:%.*]] = load <2 x i32>, ptr [[GEP_B]], align 8
+; CHECK-NEXT:    [[RED_NEXT]] = add <2 x i32> [[RED]], [[L_B]]
+; CHECK-NEXT:    [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
+; CHECK-NEXT:    [[INNER_EC:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], [[M]]
+; CHECK-NEXT:    br i1 [[INNER_EC]], label %[[OUTER_LATCH]], label %[[INNER]]
+; CHECK:       [[OUTER_LATCH]]:
+; CHECK-NEXT:    [[RED_LCSSA:%.*]] = phi <2 x i32> [ [[RED_NEXT]], %[[INNER]] ]
+; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds <2 x i32>, ptr [[A]], i64 [[OUTER_IV]]
+; CHECK-NEXT:    store <2 x i32> [[RED_LCSSA]], ptr [[GEP_A]], align 8
+; CHECK-NEXT:    [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
+; CHECK-NEXT:    [[OUTER_EC:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[OUTER_EC]], label %[[EXIT:.*]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %inner
+
+inner:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner ]
+  %red = phi <2 x i32> [ zeroinitializer, %outer.header ], [ %red.next, %inner ]
+  %gep.B = getelementptr inbounds <2 x i32>, ptr %B, i64 %inner.iv
+  %l.B = load <2 x i32>, ptr %gep.B, align 8
+  %red.next = add <2 x i32> %red, %l.B
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.ec = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+  %red.lcssa = phi <2 x i32> [ %red.next, %inner ]
+  %gep.A = getelementptr inbounds <2 x i32>, ptr %A, i64 %outer.iv
+  store <2 x i32> %red.lcssa, ptr %gep.A, align 8
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.ec = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+define void @store_vector_live_in(ptr noalias %A, <2 x i32> %v, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @store_vector_live_in(
+; CHECK-SAME: ptr noalias [[A:%.*]], <2 x i32> [[V:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
+; CHECK:       [[OUTER_HEADER]]:
+; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT:    br label %[[INNER:.*]]
+; CHECK:       [[INNER]]:
+; CHECK-NEXT:    [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT:    [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
+; CHECK-NEXT:    [[INNER_EC:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], [[M]]
+; CHECK-NEXT:    br i1 [[INNER_EC]], label %[[OUTER_LATCH]], label %[[INNER]]
+; CHECK:       [[OUTER_LATCH]]:
+; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds <2 x i32>, ptr [[A]], i64 [[OUTER_IV]]
+; CHECK-NEXT:    store <2 x i32> [[V]], ptr [[GEP_A]], align 8
+; CHECK-NEXT:    [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
+; CHECK-NEXT:    [[OUTER_EC:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[OUTER_EC]], label %[[EXIT:.*]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP0]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %inner
+
+inner:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner ]
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.ec = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+  %gep.A = getelementptr inbounds <2 x i32>, ptr %A, i64 %outer.iv
+  store <2 x i32> %v, ptr %gep.A, align 8
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.ec = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+define void @extractelement_live_in(ptr noalias %A, <2 x i32> %v, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @extractelement_live_in(
+; CHECK-SAME: ptr noalias [[A:%.*]], <2 x i32> [[V:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
+; CHECK:       [[OUTER_HEADER]]:
+; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT:    br label %[[INNER:.*]]
+; CHECK:       [[INNER]]:
+; CHECK-NEXT:    [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT:    [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
+; CHECK-NEXT:    [[INNER_EC:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], [[M]]
+; CHECK-NEXT:    br i1 [[INNER_EC]], label %[[OUTER_LATCH]], label %[[INNER]]
+; CHECK:       [[OUTER_LATCH]]:
+; CHECK-NEXT:    [[IDX:%.*]] = and i64 [[OUTER_IV]], 1
+; CHECK-NEXT:    [[EXT:%.*]] = extractelement <2 x i32> [[V]], i64 [[IDX]]
+; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[OUTER_IV]]
+; CHECK-NEXT:    store i32 [[EXT]], ptr [[GEP_A]], align 4
+; CHECK-NEXT:    [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
+; CHECK-NEXT:    [[OUTER_EC:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[OUTER_EC]], label %[[EXIT:.*]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP0]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %inner
+
+inner:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner ]
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.ec = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+  %idx = and i64 %outer.iv, 1
+  %ext = extractelement <2 x i32> %v, i64 %idx
+  %gep.A = getelementptr inbounds i32, ptr %A, i64 %outer.iv
+  store i32 %ext, ptr %gep.A, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.ec = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+define void @cast_from_vector_live_in(ptr noalias %A, <2 x i32> %v, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @cast_from_vector_live_in(
+; CHECK-SAME: ptr noalias [[A:%.*]], <2 x i32> [[V:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
+; CHECK:       [[OUTER_HEADER]]:
+; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT:    [[BC:%.*]] = bitcast <2 x i32> [[V]] to i64
+; CHECK-NEXT:    [[ADD:%.*]] = add i64 [[BC]], [[OUTER_IV]]
+; CHECK-NEXT:    br label %[[INNER:.*]]
+; CHECK:       [[INNER]]:
+; CHECK-NEXT:    [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT:    [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
+; CHECK-NEXT:    [[INNER_EC:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], [[M]]
+; CHECK-NEXT:    br i1 [[INNER_EC]], label %[[OUTER_LATCH]], label %[[INNER]]
+; CHECK:       [[OUTER_LATCH]]:
+; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[OUTER_IV]]
+; CHECK-NEXT:    store i64 [[ADD]], ptr [[GEP_A]], align 8
+; CHECK-NEXT:    [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
+; CHECK-NEXT:    [[OUTER_EC:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[OUTER_EC]], label %[[EXIT:.*]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP0]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %bc = bitcast <2 x i32> %v to i64
+  %add = add i64 %bc, %outer.iv
+  br label %inner
+
+inner:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner ]
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.ec = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+  %gep.A = getelementptr inbounds i64, ptr %A, i64 %outer.iv
+  store i64 %add, ptr %gep.A, align 8
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.ec = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+define void @struct_typed_load(ptr noalias %A, ptr noalias %B, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @struct_typed_load(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
+; CHECK:       [[OUTER_HEADER]]:
+; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT:    [[GEP_B:%.*]] = getelementptr inbounds { i32, i32 }, ptr [[B]], i64 [[OUTER_IV]]
+; CHECK-NEXT:    [[L_B:%.*]] = load { i32, i32 }, ptr [[GEP_B]], align 4
+; CHECK-NEXT:    [[EXT:%.*]] = extractvalue { i32, i32 } [[L_B]], 1
+; CHECK-NEXT:    br label %[[INNER:.*]]
+; CHECK:       [[INNER]]:
+; CHECK-NEXT:    [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT:    [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
+; CHECK-NEXT:    [[INNER_EC:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], [[M]]
+; CHECK-NEXT:    br i1 [[INNER_EC]], label %[[OUTER_LATCH]], label %[[INNER]]
+; CHECK:       [[OUTER_LATCH]]:
+; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[OUTER_IV]]
+; CHECK-NEXT:    store i32 [[EXT]], ptr [[GEP_A]], align 4
+; CHECK-NEXT:    [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
+; CHECK-NEXT:    [[OUTER_EC:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[OUTER_EC]], label %[[EXIT:.*]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP0]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %gep.B = getelementptr inbounds { i32, i32 }, ptr %B, i64 %outer.iv
+  %l.B = load { i32, i32 }, ptr %gep.B, align 4
+  %ext = extractvalue { i32, i32 } %l.B, 1
+  br label %inner
+
+inner:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner ]
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.ec = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+  %gep.A = getelementptr inbounds i32, ptr %A, i64 %outer.iv
+  store i32 %ext, ptr %gep.A, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.ec = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+define void @struct_returning_intrinsic(ptr noalias %A, i32 %x, i64 %N, i64 %M) {
+; CHECK-LABEL: define void @struct_returning_intrinsic(
+; CHECK-SAME: ptr noalias [[A:%.*]], i32 [[X:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
+; CHECK:       [[OUTER_HEADER]]:
+; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT:    br label %[[INNER:.*]]
+; CHECK:       [[INNER]]:
+; CHECK-NEXT:    [[INNER_IV:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[INNER_IV_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT:    [[INNER_IV_NEXT]] = add nuw nsw i64 [[INNER_IV]], 1
+; CHECK-NEXT:    [[INNER_EC:%.*]] = icmp eq i64 [[INNER_IV_NEXT]], [[M]]
+; CHECK-NEXT:    br i1 [[INNER_EC]], label %[[OUTER_LATCH]], label %[[INNER]]
+; CHECK:       [[OUTER_LATCH]]:
+; CHECK-NEXT:    [[TRUNC:%.*]] = trunc i64 [[OUTER_IV]] to i32
+; CHECK-NEXT:    [[SADD:%.*]] = call { i32, i1 } @llvm.sadd.with.overflow.i32(i32 [[TRUNC]], i32 [[X]])
+; CHECK-NEXT:    [[EXT:%.*]] = extractvalue { i32, i1 } [[SADD]], 0
+; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[OUTER_IV]]
+; CHECK-NEXT:    store i32 [[EXT]], ptr [[GEP_A]], align 4
+; CHECK-NEXT:    [[OUTER_IV_NEXT]] = add nuw nsw i64 [[OUTER_IV]], 1
+; CHECK-NEXT:    [[OUTER_EC:%.*]] = icmp eq i64 [[OUTER_IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[OUTER_EC]], label %[[EXIT:.*]], label %[[OUTER_HEADER]], !llvm.loop [[LOOP0]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %inner
+
+inner:
+  %inner.iv = phi i64 [ 0, %outer.header ], [ %inner.iv.next, %inner ]
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.ec = icmp eq i64 %inner.iv.next, %M
+  br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+  %trunc = trunc i64 %outer.iv to i32
+  %sadd = call { i32, i1 } @llvm.sadd.with.overflow.i32(i32 %trunc, i32 %x)
+  %ext = extractvalue { i32, i1 } %sadd, 0
+  %gep.A = getelementptr inbounds i32, ptr %A, i64 %outer.iv
+  store i32 %ext, ptr %gep.A, align 4
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.ec = icmp eq i64 %outer.iv.next, %N
+  br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.vectorize.enable"}



More information about the llvm-commits mailing list